diff --git a/checkpoint-100/config.json b/checkpoint-100/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-100/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-100/generation_config.json b/checkpoint-100/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-100/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-100/model.safetensors b/checkpoint-100/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-100/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-100/optimizer.pt b/checkpoint-100/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..c7d72ccf12799edf93ccb5a328d91a31e2a049e9 --- /dev/null +++ b/checkpoint-100/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0e17dc36b91710af38919a82d28733c78849a9938e64d50b056fde953e08af55 +size 13823 diff --git a/checkpoint-100/rng_state.pth b/checkpoint-100/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..b9f753a9e6e82ed58582c97aeb6ca7b9a321e388 --- /dev/null +++ b/checkpoint-100/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:656c952bc98f1ba6483f4b602ab79d3a7eb64d231d7b2b6ae517f06e7e137155 +size 14455 diff --git a/checkpoint-100/scheduler.pt b/checkpoint-100/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..3069470f3b24b2dae82853458b5814e4da2ff029 --- /dev/null +++ b/checkpoint-100/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2a487ad8594fd184a1dbb5f7128f07682e4d70038d880a5d45dd29b502807a0b +size 1465 diff --git a/checkpoint-100/trainer_state.json b/checkpoint-100/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4b71df554cec5f4dfba121c1e5a9b0bdd48c3b35 --- /dev/null +++ b/checkpoint-100/trainer_state.json @@ -0,0 +1,104 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 12.5, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 675648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-100/training_args.bin b/checkpoint-100/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-100/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1000/config.json b/checkpoint-1000/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1000/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1000/generation_config.json b/checkpoint-1000/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1000/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1000/model.safetensors b/checkpoint-1000/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1000/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1000/optimizer.pt b/checkpoint-1000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..4aa3c37276e79256ad94121bf1d322649d4b9ed7 --- /dev/null +++ b/checkpoint-1000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4683cfa458233e00c198c54038d450bfd06ca52f719705e01fc34a4845b539a2 +size 13823 diff --git a/checkpoint-1000/rng_state.pth b/checkpoint-1000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..0760e590ae4cc65e582d32493f30dcec18dfdd3b --- /dev/null +++ b/checkpoint-1000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c92683043c9a8610fa78e10e63d70e47ebd8152c60d1cab4d893b74a45bb5db4 +size 14455 diff --git a/checkpoint-1000/scheduler.pt b/checkpoint-1000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..0b969d589ca83b80c52789223f116cc30fa6a9e4 --- /dev/null +++ b/checkpoint-1000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:433f845f661e9be47ccaee189de347cf46b20ad6176b2cfd945b4c290cad9fc8 +size 1465 diff --git a/checkpoint-1000/trainer_state.json b/checkpoint-1000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0b7d9ed9fbd21533b2a8af5f69358d7cb12fc88e --- /dev/null +++ b/checkpoint-1000/trainer_state.json @@ -0,0 +1,734 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 125.0, + "eval_steps": 500, + "global_step": 1000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6750000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1000/training_args.bin b/checkpoint-1000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1050/config.json b/checkpoint-1050/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1050/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1050/generation_config.json b/checkpoint-1050/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1050/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1050/model.safetensors b/checkpoint-1050/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1050/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1050/optimizer.pt b/checkpoint-1050/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..44ea31ddd5a0b5f62ca76a9691d1d00ab105d755 --- /dev/null +++ b/checkpoint-1050/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0587ee5489e3230570b30b2c6399ad8da6204af076e1b7662ccdb717d4f61c53 +size 13823 diff --git a/checkpoint-1050/rng_state.pth b/checkpoint-1050/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..aa9c052641efcd8347fe223738806517b7458634 --- /dev/null +++ b/checkpoint-1050/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:72dedc896a3c9245030a09cd03b9a14e0fdca07e3a751912053a18a37a5e6782 +size 14455 diff --git a/checkpoint-1050/scheduler.pt b/checkpoint-1050/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..fdd4db350a68f0befca362d9835e84f0e5a827f2 --- /dev/null +++ b/checkpoint-1050/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:82fa2536fa054f9bf5776f9ff6684daf9790146183edb054be6762da536dacb2 +size 1465 diff --git a/checkpoint-1050/trainer_state.json b/checkpoint-1050/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ed09fd6532356e99e48593f286ba302c70e28176 --- /dev/null +++ b/checkpoint-1050/trainer_state.json @@ -0,0 +1,769 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 131.25, + "eval_steps": 500, + "global_step": 1050, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7087824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1050/training_args.bin b/checkpoint-1050/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1050/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1100/config.json b/checkpoint-1100/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1100/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1100/generation_config.json b/checkpoint-1100/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1100/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1100/model.safetensors b/checkpoint-1100/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1100/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1100/optimizer.pt b/checkpoint-1100/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a4255da0d588a02655a0bb050495035e9a17f88a --- /dev/null +++ b/checkpoint-1100/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cecfcf884ab0004c59872c46381bfe51f9a073a67de02523a5acf9692d5ba866 +size 13823 diff --git a/checkpoint-1100/rng_state.pth b/checkpoint-1100/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..54d5f24be9ee7c123d56c8c27194130015300c3d --- /dev/null +++ b/checkpoint-1100/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5fc2de446fe61209ed367ff94c7ffb565e5e69564436f15adf49d20829abf178 +size 14455 diff --git a/checkpoint-1100/scheduler.pt b/checkpoint-1100/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..b94b14614c5fd79fd048b3597a8a16e8e59e7eee --- /dev/null +++ b/checkpoint-1100/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cb7541fe30b75a402f49ee32b1224b2f0f71b7e0c5a8d479bb8a263674caa2b4 +size 1465 diff --git a/checkpoint-1100/trainer_state.json b/checkpoint-1100/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9e2857aa3fd3153a336a2b19892a51573d3b8ae8 --- /dev/null +++ b/checkpoint-1100/trainer_state.json @@ -0,0 +1,804 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 137.5, + "eval_steps": 500, + "global_step": 1100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7425648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1100/training_args.bin b/checkpoint-1100/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1100/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1150/config.json b/checkpoint-1150/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1150/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1150/generation_config.json b/checkpoint-1150/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1150/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1150/model.safetensors b/checkpoint-1150/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1150/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1150/optimizer.pt b/checkpoint-1150/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..28a00f0729ee22bb8bd50ae5159fd49e6ecefbef --- /dev/null +++ b/checkpoint-1150/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4c6f07608caca2628e4e7d4f07ad180212b4eccace16f6c14e1ba4b9c0752cd6 +size 13823 diff --git a/checkpoint-1150/rng_state.pth b/checkpoint-1150/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..7f056ea4a10ccf22b1d60e8f434b63f4989842ea --- /dev/null +++ b/checkpoint-1150/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:adc2989b41697bc86300e49c8243db2e3b464f9053a52911e9b2ba6e76a2eee9 +size 14455 diff --git a/checkpoint-1150/scheduler.pt b/checkpoint-1150/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..798fd8f1e24f462690083b7f30f34558f1fd3ad7 --- /dev/null +++ b/checkpoint-1150/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5d053bc3befa0c54cb32d83b5ef57ea57d883adf0189ecfb06b92af8c072919c +size 1465 diff --git a/checkpoint-1150/trainer_state.json b/checkpoint-1150/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9146c875ab447ba31ff0321535e87436e840ffc2 --- /dev/null +++ b/checkpoint-1150/trainer_state.json @@ -0,0 +1,839 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 143.75, + "eval_steps": 500, + "global_step": 1150, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7763472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1150/training_args.bin b/checkpoint-1150/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1150/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1200/config.json b/checkpoint-1200/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1200/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1200/generation_config.json b/checkpoint-1200/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1200/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1200/model.safetensors b/checkpoint-1200/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1200/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1200/optimizer.pt b/checkpoint-1200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..379185ec103613b4768fe86abd92a50c5b800d70 --- /dev/null +++ b/checkpoint-1200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3975bc7a21f5a4fd5059daec4f18dc973217774caa4f4484f1069b0e3cf0034e +size 13823 diff --git a/checkpoint-1200/rng_state.pth b/checkpoint-1200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..840ff01d52ea9c15aa272a589cbf1ac919025d4a --- /dev/null +++ b/checkpoint-1200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5992258b0e831b29c37beeefed96237fb5573ddb793970294c8b6f2dc3098fd8 +size 14455 diff --git a/checkpoint-1200/scheduler.pt b/checkpoint-1200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..68ff8806cd41e31ef5966e9d0521491600e8d548 --- /dev/null +++ b/checkpoint-1200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:822b71f69a7890e9b8716f7f8e72637712c24b0c354b6f54e6d112260cd7264c +size 1465 diff --git a/checkpoint-1200/trainer_state.json b/checkpoint-1200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..232cb7d28efac2ae2ff1a449847f19b919141471 --- /dev/null +++ b/checkpoint-1200/trainer_state.json @@ -0,0 +1,874 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 150.0, + "eval_steps": 500, + "global_step": 1200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8100000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1200/training_args.bin b/checkpoint-1200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1250/config.json b/checkpoint-1250/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1250/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1250/generation_config.json b/checkpoint-1250/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1250/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1250/model.safetensors b/checkpoint-1250/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1250/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1250/optimizer.pt b/checkpoint-1250/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..aa40ad1792a2de689cd1d3b112d27a0080b1b93c --- /dev/null +++ b/checkpoint-1250/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0e98914e8bb7ddbe52f2af1a3a8d8c6859232685f3c2e43970a94383df89f297 +size 13823 diff --git a/checkpoint-1250/rng_state.pth b/checkpoint-1250/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..8fb11e0c34c171ec236931294ff1df8ac09a0bde --- /dev/null +++ b/checkpoint-1250/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d45985fd35caf3f68b22c9de463ff80a4a4049f0d86900b0fc0a73450b4fe760 +size 14455 diff --git a/checkpoint-1250/scheduler.pt b/checkpoint-1250/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..8da99f129433cb46e50be8f9f68eb0ab3a669cb1 --- /dev/null +++ b/checkpoint-1250/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:93a2970dd959e17c1748a5b87ce5126348f55ed0c23baa2efa8b6b94b94b7ea2 +size 1465 diff --git a/checkpoint-1250/trainer_state.json b/checkpoint-1250/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9f0ef16b61957d8c8fd4e5ea5d2901d57eaf823e --- /dev/null +++ b/checkpoint-1250/trainer_state.json @@ -0,0 +1,909 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 156.25, + "eval_steps": 500, + "global_step": 1250, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8437824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1250/training_args.bin b/checkpoint-1250/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1250/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1300/config.json b/checkpoint-1300/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1300/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1300/generation_config.json b/checkpoint-1300/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1300/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1300/model.safetensors b/checkpoint-1300/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1300/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1300/optimizer.pt b/checkpoint-1300/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..24324bf8d88d911b2969ef1ef01b66119e0721ba --- /dev/null +++ b/checkpoint-1300/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:43d60a750e9501419f6aa99634eb7602dccbd430216fb44a5b3226c5d2a6c8b5 +size 13823 diff --git a/checkpoint-1300/rng_state.pth b/checkpoint-1300/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..5e2a98b34eedbb912d9ebd1c06bf6eb374deead8 --- /dev/null +++ b/checkpoint-1300/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:320cb05eafd32085262f2a246a0b78a4d61864095818131fcbf54d3a3f398db7 +size 14455 diff --git a/checkpoint-1300/scheduler.pt b/checkpoint-1300/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..fdf66c02fdd2dba6c8f44b7238425d08f4b8f422 --- /dev/null +++ b/checkpoint-1300/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:01f29c4c01cbc1ada7ccb317b214fbd045ff81a460889cac44ab379952f7e054 +size 1465 diff --git a/checkpoint-1300/trainer_state.json b/checkpoint-1300/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..8414c5b639ecf2049dc2fdcb1a2f6c5d8219e68a --- /dev/null +++ b/checkpoint-1300/trainer_state.json @@ -0,0 +1,944 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 162.5, + "eval_steps": 500, + "global_step": 1300, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8775648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1300/training_args.bin b/checkpoint-1300/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1300/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1350/config.json b/checkpoint-1350/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1350/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1350/generation_config.json b/checkpoint-1350/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1350/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1350/model.safetensors b/checkpoint-1350/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1350/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1350/optimizer.pt b/checkpoint-1350/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..7f8d53673a29b0de80e28c8c63a4fe43998c442a --- /dev/null +++ b/checkpoint-1350/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b30d7b3da78e2907ba42c7d6a23ac30667ee352d5afbaeca2d0d5ff1451c3025 +size 13823 diff --git a/checkpoint-1350/rng_state.pth b/checkpoint-1350/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..0f5e9e3723ebaa3e3a9d859537d8c0d6799dc42f --- /dev/null +++ b/checkpoint-1350/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fad86adae2a5706eded205a4fa67f74014440e4cb0adf94faf51bd748b8c509a +size 14455 diff --git a/checkpoint-1350/scheduler.pt b/checkpoint-1350/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..32432122d199d791c9fe955e7d0a74a352c4dac5 --- /dev/null +++ b/checkpoint-1350/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:765feaf60a91f27081f9f618893b1d9f34f9b5323a12009adb4d97d9c1e3f77e +size 1465 diff --git a/checkpoint-1350/trainer_state.json b/checkpoint-1350/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..82493adc774e80b847f0bafeeb4264e0f75ca50d --- /dev/null +++ b/checkpoint-1350/trainer_state.json @@ -0,0 +1,979 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 168.75, + "eval_steps": 500, + "global_step": 1350, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9113472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1350/training_args.bin b/checkpoint-1350/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1350/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1400/config.json b/checkpoint-1400/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1400/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1400/generation_config.json b/checkpoint-1400/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1400/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1400/model.safetensors b/checkpoint-1400/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1400/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1400/optimizer.pt b/checkpoint-1400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..aff3ca2890f70f539c766dce3e05909c8bc562be --- /dev/null +++ b/checkpoint-1400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d9bab42157c54f8ade40f8d1a0cb1bc7cf08eb784db147ec7a08c49fdf0204e6 +size 13823 diff --git a/checkpoint-1400/rng_state.pth b/checkpoint-1400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..dd22c3ae6fca4405eac704db73086e628fc20827 --- /dev/null +++ b/checkpoint-1400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:83eeb3e5644ff8f4a860672e691b9a9c8596b6f8fd0ade46ae96f97103ab215f +size 14455 diff --git a/checkpoint-1400/scheduler.pt b/checkpoint-1400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..fa6b053c773a0dbd9957aee9ed31496a85e5dd35 --- /dev/null +++ b/checkpoint-1400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d597ceb33f66d7217b0c796f4882c0dba62f97fde0f622cdc910598f578bd589 +size 1465 diff --git a/checkpoint-1400/trainer_state.json b/checkpoint-1400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6a981787ee8896867109c3ac430badfe95d8e14a --- /dev/null +++ b/checkpoint-1400/trainer_state.json @@ -0,0 +1,1014 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 175.0, + "eval_steps": 500, + "global_step": 1400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9450000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1400/training_args.bin b/checkpoint-1400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1450/config.json b/checkpoint-1450/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1450/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1450/generation_config.json b/checkpoint-1450/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1450/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1450/model.safetensors b/checkpoint-1450/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1450/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1450/optimizer.pt b/checkpoint-1450/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..0c995399a4b93629ab585cabc9a76e1f910b6df9 --- /dev/null +++ b/checkpoint-1450/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:541f6ff29a8021b6084c1c33c583c8b22dad945d7b098f2cec6f15551c04eed3 +size 13823 diff --git a/checkpoint-1450/rng_state.pth b/checkpoint-1450/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..47eab4039b07e96f899edb7b0618652033225e5f --- /dev/null +++ b/checkpoint-1450/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e519cb09a72cc913ceaec702bad58eea5b84d437a6eb10cba117e48514bc2a89 +size 14455 diff --git a/checkpoint-1450/scheduler.pt b/checkpoint-1450/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..7ff66cd4f21a64566e3662d75eff094933dd6433 --- /dev/null +++ b/checkpoint-1450/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:495a5af2274e4388eda7ba37d2b297fdace030c04ca6fbdd6489245fcda65990 +size 1465 diff --git a/checkpoint-1450/trainer_state.json b/checkpoint-1450/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9eab5cbf961d3f66a4c13b141234e316dbf22279 --- /dev/null +++ b/checkpoint-1450/trainer_state.json @@ -0,0 +1,1049 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 181.25, + "eval_steps": 500, + "global_step": 1450, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9787824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1450/training_args.bin b/checkpoint-1450/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1450/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-150/config.json b/checkpoint-150/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-150/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-150/generation_config.json b/checkpoint-150/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-150/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-150/model.safetensors b/checkpoint-150/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-150/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-150/optimizer.pt b/checkpoint-150/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a90e683dffa286f4828a1fd5bc274eefe1d58443 --- /dev/null +++ b/checkpoint-150/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0e2feedeeb47ab9d26ebd24a3cb5dcb06650cd77053b3a3b7eeceafa4fdc1932 +size 13823 diff --git a/checkpoint-150/rng_state.pth b/checkpoint-150/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..5a93c77247abec2812a13bf70a70c37d87c71e09 --- /dev/null +++ b/checkpoint-150/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b9b5fdf389d8144a43074a17a83045fcaacf31701f4eb93ca74497b6fb1054c0 +size 14455 diff --git a/checkpoint-150/scheduler.pt b/checkpoint-150/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..47b62a998ff9482240149feba104f49f85253b7e --- /dev/null +++ b/checkpoint-150/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fb2d5e876b26b4f2e412f73b40b068dd2196fd7a351e34fbb6611fd253cd8108 +size 1465 diff --git a/checkpoint-150/trainer_state.json b/checkpoint-150/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..41d7eb001ed4a031a0ccc06f598295ca6823b75a --- /dev/null +++ b/checkpoint-150/trainer_state.json @@ -0,0 +1,139 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 18.75, + "eval_steps": 500, + "global_step": 150, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1013472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-150/training_args.bin b/checkpoint-150/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-150/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1500/config.json b/checkpoint-1500/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1500/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1500/generation_config.json b/checkpoint-1500/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1500/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1500/model.safetensors b/checkpoint-1500/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1500/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1500/optimizer.pt b/checkpoint-1500/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..ff8eaded8ae6de3be346e08b02ae7692f06f101e --- /dev/null +++ b/checkpoint-1500/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e0deb1c66a476ef62899537b41600b8875bf2033803c153f556f6fd944ee9875 +size 13823 diff --git a/checkpoint-1500/rng_state.pth b/checkpoint-1500/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..0c27953757af1d04b232d4ced10c5bb8ab76e80d --- /dev/null +++ b/checkpoint-1500/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2471408a1d4731e70a9d36dd3cb61c5d9c796394a6b103900c2f296f099273ab +size 14455 diff --git a/checkpoint-1500/scheduler.pt b/checkpoint-1500/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bb98b874c87c57764106d6980b4ffc83ad090dd1 --- /dev/null +++ b/checkpoint-1500/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1c69d62a64c88d98b88f66d4f37eb5f2bca61557660db939a38ca3fe78af810a +size 1465 diff --git a/checkpoint-1500/trainer_state.json b/checkpoint-1500/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..89c20e07a6d474ff0e9d85e4b405e5f808c589b8 --- /dev/null +++ b/checkpoint-1500/trainer_state.json @@ -0,0 +1,1084 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 187.5, + "eval_steps": 500, + "global_step": 1500, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 10125648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1500/training_args.bin b/checkpoint-1500/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1500/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1550/config.json b/checkpoint-1550/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1550/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1550/generation_config.json b/checkpoint-1550/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1550/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1550/model.safetensors b/checkpoint-1550/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1550/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1550/optimizer.pt b/checkpoint-1550/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..188a5083ef08f21842c93a63114440de4efcd37a --- /dev/null +++ b/checkpoint-1550/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b8cce817d01e0c0ce283f17e1201eaeff123283dae580660491355431e15c8f2 +size 13823 diff --git a/checkpoint-1550/rng_state.pth b/checkpoint-1550/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..7f8cd24e60012b9ce499159e37f3da204d1e728a --- /dev/null +++ b/checkpoint-1550/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15f90592b84670b1166a1734757754eec3bd52fb7ca4c3a73e15259b6256c145 +size 14455 diff --git a/checkpoint-1550/scheduler.pt b/checkpoint-1550/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bc3df6e8635382e3ac5c8f7dc7514d664c2397f5 --- /dev/null +++ b/checkpoint-1550/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d7b0638b101affeab5643f82b6496586035c93c201c6177b5ef1e6be8ef9673b +size 1465 diff --git a/checkpoint-1550/trainer_state.json b/checkpoint-1550/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1ea0e5ea334bffee099ee604e111f1d54db2ae0b --- /dev/null +++ b/checkpoint-1550/trainer_state.json @@ -0,0 +1,1119 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 193.75, + "eval_steps": 500, + "global_step": 1550, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 10463472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1550/training_args.bin b/checkpoint-1550/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1550/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1600/config.json b/checkpoint-1600/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1600/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1600/generation_config.json b/checkpoint-1600/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1600/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1600/model.safetensors b/checkpoint-1600/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1600/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1600/optimizer.pt b/checkpoint-1600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f13d15d226e7b952db28b46ad805ef1faf529845 --- /dev/null +++ b/checkpoint-1600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c11f4cfc713166a5c966ddf0b8c5e985df728c7497e31a5d4cc7dfa139e62535 +size 13823 diff --git a/checkpoint-1600/rng_state.pth b/checkpoint-1600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..3c7b3b0c11639052c79541d8bd5b16e54436f192 --- /dev/null +++ b/checkpoint-1600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9d3e7f284e9e91167f1e022bf19e2222e8af22dc14da1e18b638db0e5b4f7b6f +size 14455 diff --git a/checkpoint-1600/scheduler.pt b/checkpoint-1600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..56852eb07f752710000d3a1dfc7f409da6f8f9c5 --- /dev/null +++ b/checkpoint-1600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d1fd5dc89a152e572445b9cd352beb43ce4ed60f374cadd0adc6cab0be748390 +size 1465 diff --git a/checkpoint-1600/trainer_state.json b/checkpoint-1600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..838729ae681667b41f57e80c86ba7471a28af9d1 --- /dev/null +++ b/checkpoint-1600/trainer_state.json @@ -0,0 +1,1154 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 200.0, + "eval_steps": 500, + "global_step": 1600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 10800000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1600/training_args.bin b/checkpoint-1600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1650/config.json b/checkpoint-1650/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1650/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1650/generation_config.json b/checkpoint-1650/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1650/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1650/model.safetensors b/checkpoint-1650/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1650/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1650/optimizer.pt b/checkpoint-1650/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d3a0928b23784b09bc5daeb58568c621795319f2 --- /dev/null +++ b/checkpoint-1650/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fbd6c58235fffc246aa33fd21f1e41c40d71b539f8172c5c2f6e5e2c27d56bb3 +size 13823 diff --git a/checkpoint-1650/rng_state.pth b/checkpoint-1650/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..59440b3891fd9cac1d33964a621a50d69ed54dfe --- /dev/null +++ b/checkpoint-1650/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3535ded395c6328e59af5d4f5f22c580906700185db9baa9a1001bbe9d0719c1 +size 14455 diff --git a/checkpoint-1650/scheduler.pt b/checkpoint-1650/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..f1cd7b522fdafbdf1096025e1d3378f9c67e5755 --- /dev/null +++ b/checkpoint-1650/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:526450179f7e4deb73bdf325354464f8d111e13c87e7b878c7cf5f400499ae05 +size 1465 diff --git a/checkpoint-1650/trainer_state.json b/checkpoint-1650/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ae43d66202730df915ac7f9de9fcfcddcc9f1023 --- /dev/null +++ b/checkpoint-1650/trainer_state.json @@ -0,0 +1,1189 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 206.25, + "eval_steps": 500, + "global_step": 1650, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 11137824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1650/training_args.bin b/checkpoint-1650/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1650/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1700/config.json b/checkpoint-1700/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1700/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1700/generation_config.json b/checkpoint-1700/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1700/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1700/model.safetensors b/checkpoint-1700/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1700/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1700/optimizer.pt b/checkpoint-1700/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d975335711efc18fe2af0180674a93bc0f23e000 --- /dev/null +++ b/checkpoint-1700/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9ca8f3c52c58dcb576ee8a261cfc6494d65ffd60d35e2ebb7e8b6ae84e9c3aab +size 13823 diff --git a/checkpoint-1700/rng_state.pth b/checkpoint-1700/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..3ee1e1f79d16171eec89c548245670636b06c70a --- /dev/null +++ b/checkpoint-1700/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2e8ed2248d23ee63a6609e5e947adf050fd8af4016f606a5afe97b6b68e7adab +size 14455 diff --git a/checkpoint-1700/scheduler.pt b/checkpoint-1700/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..691d5b2217ea85d2a2d65b99387a3f06188d9778 --- /dev/null +++ b/checkpoint-1700/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a0bbd6c10fc11e6ed16a39daf17aa3dfafa96b96e49081d4d3c00b1bccd06486 +size 1465 diff --git a/checkpoint-1700/trainer_state.json b/checkpoint-1700/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..989d51d02bd3cd74b1fb88dc697580fde2d0a2ae --- /dev/null +++ b/checkpoint-1700/trainer_state.json @@ -0,0 +1,1224 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 212.5, + "eval_steps": 500, + "global_step": 1700, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 11475648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1700/training_args.bin b/checkpoint-1700/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1700/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1750/config.json b/checkpoint-1750/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1750/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1750/generation_config.json b/checkpoint-1750/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1750/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1750/model.safetensors b/checkpoint-1750/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1750/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1750/optimizer.pt b/checkpoint-1750/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b4c0ac021a5ee761601e72a80ac4df63ff440995 --- /dev/null +++ b/checkpoint-1750/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1aa94f28f5508d41fe3470b9f7e8e23e893600be3e4df81f8d5d116a9640b752 +size 13823 diff --git a/checkpoint-1750/rng_state.pth b/checkpoint-1750/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..c6b191b4d04c36884f7eedc4ef6fffc4afdc8fcd --- /dev/null +++ b/checkpoint-1750/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1fd1a66188e342e7a03d3c8be2b0343f2c1cb7f754634fe1aa434970e9a7892a +size 14455 diff --git a/checkpoint-1750/scheduler.pt b/checkpoint-1750/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..5050233a20fb35813863d2650dd45826912c0929 --- /dev/null +++ b/checkpoint-1750/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5ebbc29a63f1b67d5c99a8a9cc1903eafc0ccdfdf67b7ef2b9180e29f5b133ac +size 1465 diff --git a/checkpoint-1750/trainer_state.json b/checkpoint-1750/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..76e102aa6f36fd9ef4fffbf9c947c14772f60190 --- /dev/null +++ b/checkpoint-1750/trainer_state.json @@ -0,0 +1,1259 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 218.75, + "eval_steps": 500, + "global_step": 1750, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 11813472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1750/training_args.bin b/checkpoint-1750/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1750/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1800/config.json b/checkpoint-1800/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1800/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1800/generation_config.json b/checkpoint-1800/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1800/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1800/model.safetensors b/checkpoint-1800/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1800/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1800/optimizer.pt b/checkpoint-1800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f75cea3beaabf5d9f34e89a28d7fe48895128a48 --- /dev/null +++ b/checkpoint-1800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6727836a2a382e540fc81f94c2051251aa04d4fe641df11c9070440f2fea292b +size 13823 diff --git a/checkpoint-1800/rng_state.pth b/checkpoint-1800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..b350905202733b99f0b57652e5527361b00a283d --- /dev/null +++ b/checkpoint-1800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:364cfdda8878b3b970d298f51f628e7fad67369dba5c2c0b6227fbca0fda5730 +size 14455 diff --git a/checkpoint-1800/scheduler.pt b/checkpoint-1800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..989871cf3ad865a8c669b738b6326eb195c9bb3e --- /dev/null +++ b/checkpoint-1800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:318f621d67330a4a21e7b098d4848f4c4a90aa52dbbb4828bc2e9a5cca820472 +size 1465 diff --git a/checkpoint-1800/trainer_state.json b/checkpoint-1800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..887ff8f898033827d872d0f06fe5099297473011 --- /dev/null +++ b/checkpoint-1800/trainer_state.json @@ -0,0 +1,1294 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 225.0, + "eval_steps": 500, + "global_step": 1800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 12150000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1800/training_args.bin b/checkpoint-1800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1850/config.json b/checkpoint-1850/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1850/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1850/generation_config.json b/checkpoint-1850/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1850/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1850/model.safetensors b/checkpoint-1850/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1850/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1850/optimizer.pt b/checkpoint-1850/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..7e9e79299b4c67c760995e8322fbccc21c35ca33 --- /dev/null +++ b/checkpoint-1850/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e437df8bc0b5db1177ad8421ff2fb7006aa61fb9a8fee7111f99121aab04d35e +size 13823 diff --git a/checkpoint-1850/rng_state.pth b/checkpoint-1850/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9298f48805f706e77c7c06ce5874c5244692507a --- /dev/null +++ b/checkpoint-1850/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bde805bffa9efc3d4a8a84a731e59f9c9964667a275919b172b2d48e7148b9b0 +size 14455 diff --git a/checkpoint-1850/scheduler.pt b/checkpoint-1850/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..3545b9b97a3dcd614c56f7d1851dea812a0d342b --- /dev/null +++ b/checkpoint-1850/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:56c839241c5167d4bb672491cc4d3d2d17ac21aef43a65b67595efa059f6942d +size 1465 diff --git a/checkpoint-1850/trainer_state.json b/checkpoint-1850/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..72d7587890e50f3f5a9d52a93719b95a2bc10cf3 --- /dev/null +++ b/checkpoint-1850/trainer_state.json @@ -0,0 +1,1329 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 231.25, + "eval_steps": 500, + "global_step": 1850, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 12487824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1850/training_args.bin b/checkpoint-1850/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1850/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1900/config.json b/checkpoint-1900/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1900/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1900/generation_config.json b/checkpoint-1900/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1900/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1900/model.safetensors b/checkpoint-1900/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1900/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1900/optimizer.pt b/checkpoint-1900/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..9fcf576749f1d7e8a1bd09369d88c4f2ac15407a --- /dev/null +++ b/checkpoint-1900/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce51ba42f304e2005443831214701d75c2d261bd7276c92d62fb91834020e0c0 +size 13823 diff --git a/checkpoint-1900/rng_state.pth b/checkpoint-1900/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4d998987c9b04817191e3eece3dc42e3aae6bf38 --- /dev/null +++ b/checkpoint-1900/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:46bccde235ecd8474dd4715cd69b159d0e5c9cd8d0267adb8bba655190d5e1c7 +size 14455 diff --git a/checkpoint-1900/scheduler.pt b/checkpoint-1900/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d310f4a8109ddbdd56c934a210722f2fa06f43b4 --- /dev/null +++ b/checkpoint-1900/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9a817e07ca1bbefb9bceec0a405fd6a82f43a371883bc10f5d6ea5d3363e467d +size 1465 diff --git a/checkpoint-1900/trainer_state.json b/checkpoint-1900/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..58a372c0cb0cbb94193b62d5bd2e15cfea543eb8 --- /dev/null +++ b/checkpoint-1900/trainer_state.json @@ -0,0 +1,1364 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 237.5, + "eval_steps": 500, + "global_step": 1900, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 12825648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1900/training_args.bin b/checkpoint-1900/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1900/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-1950/config.json b/checkpoint-1950/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-1950/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-1950/generation_config.json b/checkpoint-1950/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-1950/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-1950/model.safetensors b/checkpoint-1950/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-1950/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-1950/optimizer.pt b/checkpoint-1950/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..9525687c96aa19acc476cb11389bc589ffa6ab94 --- /dev/null +++ b/checkpoint-1950/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0d254f86eb8dab66d5495a2ae99277c1ba997b9cd31d745bf8920bcb627a3c60 +size 13823 diff --git a/checkpoint-1950/rng_state.pth b/checkpoint-1950/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9ec9f910cbd6c87c4ecc57d4df19c339fb33c361 --- /dev/null +++ b/checkpoint-1950/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c1adfca6c2341b365997ee831f12a3e645be86cd608a814a698ff5dc8cdaa03a +size 14455 diff --git a/checkpoint-1950/scheduler.pt b/checkpoint-1950/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d591cb86ff5354e8c0b55c8dcfe6268fd69e7536 --- /dev/null +++ b/checkpoint-1950/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:827173df36b02d1f21c6b0de7559736a034ae9b34cf5165b725556f5f7e415f5 +size 1465 diff --git a/checkpoint-1950/trainer_state.json b/checkpoint-1950/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..245b4aa733a35228c55870da2bf4f8bebe4793bb --- /dev/null +++ b/checkpoint-1950/trainer_state.json @@ -0,0 +1,1399 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 243.75, + "eval_steps": 500, + "global_step": 1950, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 13163472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1950/training_args.bin b/checkpoint-1950/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-1950/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-200/config.json b/checkpoint-200/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-200/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-200/generation_config.json b/checkpoint-200/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-200/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-200/model.safetensors b/checkpoint-200/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-200/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-200/optimizer.pt b/checkpoint-200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..7bea8521bf9aa0b88478b050767217ba374afbda --- /dev/null +++ b/checkpoint-200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b750a0b35f351ee6af675bf6dc7935b9c543266836a5d2aaf381d425f74b03b6 +size 13823 diff --git a/checkpoint-200/rng_state.pth b/checkpoint-200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..5c2828e1607164337db2bacba69197d5c7ddd07c --- /dev/null +++ b/checkpoint-200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:578e8799d025fe13990ba869b383010b37f57725e649414cca523ec437e2196e +size 14455 diff --git a/checkpoint-200/scheduler.pt b/checkpoint-200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..99a529294272dd1db7a1db9b28d66467faaa46a4 --- /dev/null +++ b/checkpoint-200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b23106dcc0044928b3507023ef68f088389c0e042fda57d0acd6bf1fa776940a +size 1465 diff --git a/checkpoint-200/trainer_state.json b/checkpoint-200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4e34081f2443cbcf70a8d9b21e0dcff9a3c2051a --- /dev/null +++ b/checkpoint-200/trainer_state.json @@ -0,0 +1,174 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 25.0, + "eval_steps": 500, + "global_step": 200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1350000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-200/training_args.bin b/checkpoint-200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2000/config.json b/checkpoint-2000/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2000/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2000/generation_config.json b/checkpoint-2000/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2000/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2000/model.safetensors b/checkpoint-2000/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2000/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2000/optimizer.pt b/checkpoint-2000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..230ecc2995e783161a163422d59cce02109e4f52 --- /dev/null +++ b/checkpoint-2000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:317cd62420f9be4909e28c0cb554a6a3586d54ed489a37b2ba37312722687499 +size 13823 diff --git a/checkpoint-2000/rng_state.pth b/checkpoint-2000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9eee65daaf537195fac28c0a1a859124d8ff880a --- /dev/null +++ b/checkpoint-2000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d8f1d8cd0f0b19b09f618638626f5c36a487826fb6d0901d1ca38564c79eb31f +size 14455 diff --git a/checkpoint-2000/scheduler.pt b/checkpoint-2000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..88ac14aec2929aba4541d671bdf264333912fcf8 --- /dev/null +++ b/checkpoint-2000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2d619d7f64a8f87cadb5363995305c60220c3173b55dd4f132beb67300726ea9 +size 1465 diff --git a/checkpoint-2000/trainer_state.json b/checkpoint-2000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4f8a07d1272142b0f838299eeaba69b0b26c0d74 --- /dev/null +++ b/checkpoint-2000/trainer_state.json @@ -0,0 +1,1434 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 250.0, + "eval_steps": 500, + "global_step": 2000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 13500000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2000/training_args.bin b/checkpoint-2000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2050/config.json b/checkpoint-2050/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2050/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2050/generation_config.json b/checkpoint-2050/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2050/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2050/model.safetensors b/checkpoint-2050/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2050/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2050/optimizer.pt b/checkpoint-2050/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b7e25313b6b5be3f20f6befda3cc1dad61420c1d --- /dev/null +++ b/checkpoint-2050/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b2ecbb58e2bb64b171272e5d2cc386755966350f86dd4ebd1e486509926d72b9 +size 13823 diff --git a/checkpoint-2050/rng_state.pth b/checkpoint-2050/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..205b7fff4d011fecd545dbece1928d2a53904f4d --- /dev/null +++ b/checkpoint-2050/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3801dfdfa7c675d143ff843f88a67b2e775f4f313dce67248d5b715e3fd515f9 +size 14455 diff --git a/checkpoint-2050/scheduler.pt b/checkpoint-2050/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..df19ec22a24b3d00fabc2dbf2abf17b1ea896152 --- /dev/null +++ b/checkpoint-2050/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:83eca040d3b68198240d0480506b1138526df5f55dd5dada5fbc6ccd0b010672 +size 1465 diff --git a/checkpoint-2050/trainer_state.json b/checkpoint-2050/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..64161789f62a8700df8af514d0bc93ee3204645e --- /dev/null +++ b/checkpoint-2050/trainer_state.json @@ -0,0 +1,1469 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 256.25, + "eval_steps": 500, + "global_step": 2050, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 13837824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2050/training_args.bin b/checkpoint-2050/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2050/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2100/config.json b/checkpoint-2100/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2100/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2100/generation_config.json b/checkpoint-2100/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2100/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2100/model.safetensors b/checkpoint-2100/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2100/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2100/optimizer.pt b/checkpoint-2100/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b703bfe0b53642b025882cf81e01fb0d2c0101e4 --- /dev/null +++ b/checkpoint-2100/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5a9f03fde2178d81d69e3ddb3b77d53ee2a8e719f32cae4543a0008adf566df7 +size 13823 diff --git a/checkpoint-2100/rng_state.pth b/checkpoint-2100/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..a06e1fd27f5f178d9bc0ad30df460ab3576a1663 --- /dev/null +++ b/checkpoint-2100/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:34ffb7364a87a293a80fe027b50c4d2583c581c041ee9d1c8ee90bcc5a98571a +size 14455 diff --git a/checkpoint-2100/scheduler.pt b/checkpoint-2100/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bd01e3c4f9a04fd1e0176befcd2b0042f97ead2e --- /dev/null +++ b/checkpoint-2100/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:77966fefa31fde3eab08b77b80b670ad74c51617cfa677cc568812ae24af4879 +size 1465 diff --git a/checkpoint-2100/trainer_state.json b/checkpoint-2100/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..072ecb248d413488c95c0b3fcf3c9ac521e3aff8 --- /dev/null +++ b/checkpoint-2100/trainer_state.json @@ -0,0 +1,1504 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 262.5, + "eval_steps": 500, + "global_step": 2100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 14175648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2100/training_args.bin b/checkpoint-2100/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2100/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2150/config.json b/checkpoint-2150/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2150/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2150/generation_config.json b/checkpoint-2150/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2150/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2150/model.safetensors b/checkpoint-2150/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2150/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2150/optimizer.pt b/checkpoint-2150/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..2d9c0d960f7d8dec6c71f1e5df62ec9e22b377b0 --- /dev/null +++ b/checkpoint-2150/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:821f8bea147c460eeea3279818e924a3e0e6e07377c17967204d3c9383fc961e +size 13823 diff --git a/checkpoint-2150/rng_state.pth b/checkpoint-2150/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..3189d375a9ef5fecb99dc6d99d5cf9c559fad101 --- /dev/null +++ b/checkpoint-2150/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0018c0959d14b60c0f76db30e69869d2359bd70c95fe8d4e846242b5444146da +size 14455 diff --git a/checkpoint-2150/scheduler.pt b/checkpoint-2150/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..acc5a9cc2ad6a6dea1afd3ee6d3a156df98b09b2 --- /dev/null +++ b/checkpoint-2150/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8fa9ee256fa9561b481ebff9a79719644d55551fe58fc6f76afe924db981cc69 +size 1465 diff --git a/checkpoint-2150/trainer_state.json b/checkpoint-2150/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4872e2f72ccc3d00f19224de84c3c2af70e1ad25 --- /dev/null +++ b/checkpoint-2150/trainer_state.json @@ -0,0 +1,1539 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 268.75, + "eval_steps": 500, + "global_step": 2150, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 14513472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2150/training_args.bin b/checkpoint-2150/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2150/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2200/config.json b/checkpoint-2200/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2200/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2200/generation_config.json b/checkpoint-2200/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2200/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2200/model.safetensors b/checkpoint-2200/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2200/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2200/optimizer.pt b/checkpoint-2200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..81628bcbbd94e7516146800b5c4df43f2737400c --- /dev/null +++ b/checkpoint-2200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:54bc2a4730907b13d497b46521f3e7f055a287fb9c2a19acac8ee4b5110ae208 +size 13823 diff --git a/checkpoint-2200/rng_state.pth b/checkpoint-2200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..5dccb52fb62e64518b4997eeb35917535c2977fb --- /dev/null +++ b/checkpoint-2200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d09c77ab6c84157456ed5344329f7a972cb16a67bb1dd5d4ea12e96af638ee9b +size 14455 diff --git a/checkpoint-2200/scheduler.pt b/checkpoint-2200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..f8344038344d6ae60d50c471ad3fae1010004aba --- /dev/null +++ b/checkpoint-2200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10f91ad8ac8dbbe69a43f43cfd536b177b210dcb1b08f0e27f323b3017c792ea +size 1465 diff --git a/checkpoint-2200/trainer_state.json b/checkpoint-2200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..13ade9a2b83f3522460c5ad23293c1d7c44cc446 --- /dev/null +++ b/checkpoint-2200/trainer_state.json @@ -0,0 +1,1574 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 275.0, + "eval_steps": 500, + "global_step": 2200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 14850000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2200/training_args.bin b/checkpoint-2200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2250/config.json b/checkpoint-2250/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2250/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2250/generation_config.json b/checkpoint-2250/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2250/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2250/model.safetensors b/checkpoint-2250/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2250/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2250/optimizer.pt b/checkpoint-2250/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..4776c8c79e893cc8dad583f9a2f00d937a6c3569 --- /dev/null +++ b/checkpoint-2250/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:50be935d074e0c167a19feb9d258020850e70b1563e3f1e3d8c9a9ca8c3c1507 +size 13823 diff --git a/checkpoint-2250/rng_state.pth b/checkpoint-2250/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9f9c11b2bbfc36b9d9a66ddf0d1207d9991051e8 --- /dev/null +++ b/checkpoint-2250/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:486c34d3ce9cc7b3b376110739f68fc99ce322ecea4b6c55ccc0c314d4f373d0 +size 14455 diff --git a/checkpoint-2250/scheduler.pt b/checkpoint-2250/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..e961ddb87cd1b23d0c60a5fd67f0d2b941ff7e31 --- /dev/null +++ b/checkpoint-2250/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a06022935b1e222836ef2986baa09694689e2f585f0d2abf8c198e74be5e52df +size 1465 diff --git a/checkpoint-2250/trainer_state.json b/checkpoint-2250/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..dabd90891f9bfd0eef2e261821dd041e68b6d2d6 --- /dev/null +++ b/checkpoint-2250/trainer_state.json @@ -0,0 +1,1609 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 281.25, + "eval_steps": 500, + "global_step": 2250, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 15187824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2250/training_args.bin b/checkpoint-2250/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2250/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2300/config.json b/checkpoint-2300/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2300/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2300/generation_config.json b/checkpoint-2300/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2300/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2300/model.safetensors b/checkpoint-2300/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2300/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2300/optimizer.pt b/checkpoint-2300/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..9acf7c673c13bd61bc563291883876563b4025c9 --- /dev/null +++ b/checkpoint-2300/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cec558969f89bcc13bd694e842c66f052f688b01b5fae99773581c67103caaf8 +size 13823 diff --git a/checkpoint-2300/rng_state.pth b/checkpoint-2300/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9883053b5f13afbb43541132d2ad984c87bf6915 --- /dev/null +++ b/checkpoint-2300/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:661f31b1e31d28f41ecf5559c862030340d82aee63a6a488546a6888e38632ab +size 14455 diff --git a/checkpoint-2300/scheduler.pt b/checkpoint-2300/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..c2424e78f869971ff922df4cc1afb018670abd42 --- /dev/null +++ b/checkpoint-2300/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e9766aa437d1078fdee82f7a146bfa30a9fd27363bcdae60da32f5ef6b482972 +size 1465 diff --git a/checkpoint-2300/trainer_state.json b/checkpoint-2300/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..14b9f24e4bc8226461dc7c9935ef8f030b206af8 --- /dev/null +++ b/checkpoint-2300/trainer_state.json @@ -0,0 +1,1644 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 287.5, + "eval_steps": 500, + "global_step": 2300, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 15525648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2300/training_args.bin b/checkpoint-2300/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2300/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2350/config.json b/checkpoint-2350/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2350/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2350/generation_config.json b/checkpoint-2350/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2350/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2350/model.safetensors b/checkpoint-2350/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2350/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2350/optimizer.pt b/checkpoint-2350/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..3eb23f96bf0d215138532d06292fe144b30be074 --- /dev/null +++ b/checkpoint-2350/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:571f46fcdaf1068ce73ebe4e75de5ea4acd6ab259c2267f9f1e1943608908198 +size 13823 diff --git a/checkpoint-2350/rng_state.pth b/checkpoint-2350/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..23cdc7580a87d40af967c48c93ade12fc5b16219 --- /dev/null +++ b/checkpoint-2350/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8aab24924a47bf9215c8fe64fda78f226d9ce7502d77471035d77cd6154cd604 +size 14455 diff --git a/checkpoint-2350/scheduler.pt b/checkpoint-2350/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..54ed537fc40b384645b8189549764bc96e345334 --- /dev/null +++ b/checkpoint-2350/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:88c6a0b7bf6c18cad2b0bba4740044303de0673b9f8eda518fa86c6131f74666 +size 1465 diff --git a/checkpoint-2350/trainer_state.json b/checkpoint-2350/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c26c1936d03c7dfbe8f96e50f6e77c865fe9dea5 --- /dev/null +++ b/checkpoint-2350/trainer_state.json @@ -0,0 +1,1679 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 293.75, + "eval_steps": 500, + "global_step": 2350, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 15863472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2350/training_args.bin b/checkpoint-2350/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2350/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2400/config.json b/checkpoint-2400/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2400/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2400/generation_config.json b/checkpoint-2400/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2400/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2400/model.safetensors b/checkpoint-2400/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2400/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2400/optimizer.pt b/checkpoint-2400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..c2b2674ded7e7b65f53d4ea16d43ceb5872930ce --- /dev/null +++ b/checkpoint-2400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce12db5011f578d846671a64a1624ad01b4a5fb89b43f9f0b2f1a9c4fbd95c82 +size 13823 diff --git a/checkpoint-2400/rng_state.pth b/checkpoint-2400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..944593a519ddac6e44ba7b42b2cd3e7239696870 --- /dev/null +++ b/checkpoint-2400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:da1774b4074e840f12614a902b1a46df14b20e67095373379548abdaf08f6579 +size 14455 diff --git a/checkpoint-2400/scheduler.pt b/checkpoint-2400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..815bea4c3d34c7ed29f6b8fb2ba1f226385a808d --- /dev/null +++ b/checkpoint-2400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:75dc13127375e2fce1aa50c5241aa94b1c25b3e5164591007f42a9cdf5b42fd5 +size 1465 diff --git a/checkpoint-2400/trainer_state.json b/checkpoint-2400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..44f1c93c6b2f9b527a7b0ff818b8ba5a81fb912f --- /dev/null +++ b/checkpoint-2400/trainer_state.json @@ -0,0 +1,1714 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 300.0, + "eval_steps": 500, + "global_step": 2400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 16200000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2400/training_args.bin b/checkpoint-2400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2450/config.json b/checkpoint-2450/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2450/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2450/generation_config.json b/checkpoint-2450/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2450/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2450/model.safetensors b/checkpoint-2450/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2450/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2450/optimizer.pt b/checkpoint-2450/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..2356f536d7628baa7932676fda03e20d698db570 --- /dev/null +++ b/checkpoint-2450/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:df1f166e648a71da7eea566fdc290aa4a4b0e2b3c00da1ef10ef76c274543646 +size 13823 diff --git a/checkpoint-2450/rng_state.pth b/checkpoint-2450/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4b94b09cbf0530c39bec49e448b34e8912b3b960 --- /dev/null +++ b/checkpoint-2450/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:963ad2b8b62d612681dae7b65f069d5b03e9138a68b760b0af9ea3ad5eae4135 +size 14455 diff --git a/checkpoint-2450/scheduler.pt b/checkpoint-2450/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..a5b48b886b508418712c321641f2a47db6cf0169 --- /dev/null +++ b/checkpoint-2450/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f8c380a109f4ab5613a6f0138fdfc446e3a163e06f89e857d6903b3bc67aa7d8 +size 1465 diff --git a/checkpoint-2450/trainer_state.json b/checkpoint-2450/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..872592704e64ee23f0b5b7f1aa9950cc31849b77 --- /dev/null +++ b/checkpoint-2450/trainer_state.json @@ -0,0 +1,1749 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 306.25, + "eval_steps": 500, + "global_step": 2450, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 16537824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2450/training_args.bin b/checkpoint-2450/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2450/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-250/config.json b/checkpoint-250/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-250/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-250/generation_config.json b/checkpoint-250/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-250/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-250/model.safetensors b/checkpoint-250/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-250/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-250/optimizer.pt b/checkpoint-250/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..e2ca1fe8e646ba47858349155f35b06f13598cf2 --- /dev/null +++ b/checkpoint-250/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:aab3f5a21172d07a5417b8ea9064f2c35516404d4e61ce1415e8d80c19045212 +size 13823 diff --git a/checkpoint-250/rng_state.pth b/checkpoint-250/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..973db96b4903c385b4438c800a2649e80c03c61e --- /dev/null +++ b/checkpoint-250/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d93cf2a6f5b82bd08f8b51ea11b7d93fb36e5dee432e6b631b5bd5365d1aa213 +size 14455 diff --git a/checkpoint-250/scheduler.pt b/checkpoint-250/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..b956961edbde73ae7eb9e36a2bc49a6f60e12c7b --- /dev/null +++ b/checkpoint-250/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dd201438d41b92c8a02f53cebca2f72904d2a750993579cc1666194d1892f4f9 +size 1465 diff --git a/checkpoint-250/trainer_state.json b/checkpoint-250/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c586201999973b56300c75982d676db0b5efa01c --- /dev/null +++ b/checkpoint-250/trainer_state.json @@ -0,0 +1,209 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 31.25, + "eval_steps": 500, + "global_step": 250, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1687824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-250/training_args.bin b/checkpoint-250/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-250/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2500/config.json b/checkpoint-2500/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2500/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2500/generation_config.json b/checkpoint-2500/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2500/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2500/model.safetensors b/checkpoint-2500/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2500/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2500/optimizer.pt b/checkpoint-2500/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d96f4ee737524b3be07c31ea0e45a917119e1735 --- /dev/null +++ b/checkpoint-2500/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b6c0a228f64fe5e7a3c139f19742023ac8f1efbec7792eeb472a625f39c04a8b +size 13823 diff --git a/checkpoint-2500/rng_state.pth b/checkpoint-2500/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..09a860c1f6082cf7e3939a1423062a5241fdea1c --- /dev/null +++ b/checkpoint-2500/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3f971bc42859b8cc8fa22e0ffcb85918ab5f63339d196d3c74a61a3c21526362 +size 14455 diff --git a/checkpoint-2500/scheduler.pt b/checkpoint-2500/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..4e3c95366e3b4ba8d5f576ba887adda27bbd646a --- /dev/null +++ b/checkpoint-2500/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2c72123ea1e2ec044bde8e107b3da759f74011b2c9d04d322afdc921f6b7368d +size 1465 diff --git a/checkpoint-2500/trainer_state.json b/checkpoint-2500/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ad6dc9bbb044a3287465f97f0bff46fac79086b0 --- /dev/null +++ b/checkpoint-2500/trainer_state.json @@ -0,0 +1,1784 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 312.5, + "eval_steps": 500, + "global_step": 2500, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 16875648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2500/training_args.bin b/checkpoint-2500/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2500/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2550/config.json b/checkpoint-2550/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2550/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2550/generation_config.json b/checkpoint-2550/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2550/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2550/model.safetensors b/checkpoint-2550/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2550/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2550/optimizer.pt b/checkpoint-2550/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..71bb4b3e818bf1aae77e32e6fcb42a3b8a2e2673 --- /dev/null +++ b/checkpoint-2550/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:74bc65d7c4b184a15a57103ae0c68f1dc5d178de7936fec179fecbbbb1390268 +size 13823 diff --git a/checkpoint-2550/rng_state.pth b/checkpoint-2550/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..90338795f4024872940b50c05fb043e0a642b830 --- /dev/null +++ b/checkpoint-2550/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7bc3342605525c323c9a95030e7801e73fcc1f7b73192d2fbc14a1bc62c07cf6 +size 14455 diff --git a/checkpoint-2550/scheduler.pt b/checkpoint-2550/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..deb3345889c64137c4ae25311c90dda8774e0e46 --- /dev/null +++ b/checkpoint-2550/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3e511bef3fb9c8c7d0602793490a118e601589ba0d8ef94971b400b1ea4ecba3 +size 1465 diff --git a/checkpoint-2550/trainer_state.json b/checkpoint-2550/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f7bf9ffce7ca456af275f7c8d0631f242b1dadfe --- /dev/null +++ b/checkpoint-2550/trainer_state.json @@ -0,0 +1,1819 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 318.75, + "eval_steps": 500, + "global_step": 2550, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 17213472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2550/training_args.bin b/checkpoint-2550/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2550/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2600/config.json b/checkpoint-2600/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2600/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2600/generation_config.json b/checkpoint-2600/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2600/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2600/model.safetensors b/checkpoint-2600/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2600/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2600/optimizer.pt b/checkpoint-2600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..47232c8140ee315a678bdb0c9ec5fa7067df2264 --- /dev/null +++ b/checkpoint-2600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d733edd0d0ef0f07285b069437999a193edd6885c737f913f45a5235aa5e94e3 +size 13823 diff --git a/checkpoint-2600/rng_state.pth b/checkpoint-2600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9d21e4e2b695248141b11b1efcd61b17ac284f2b --- /dev/null +++ b/checkpoint-2600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6197a9b258b704cd21f02738b41456d7d7cd6e4dbd0a88225132d29ca623e949 +size 14455 diff --git a/checkpoint-2600/scheduler.pt b/checkpoint-2600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..997440acf6183e3a372b04641ffb4ca505751012 --- /dev/null +++ b/checkpoint-2600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dd7f65d17992dc17935758d6b5eb81f386161d3e2787cb72bf2f92b3c5aed5d2 +size 1465 diff --git a/checkpoint-2600/trainer_state.json b/checkpoint-2600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6f8a83507b846f946002157b3fd918a9b6da56db --- /dev/null +++ b/checkpoint-2600/trainer_state.json @@ -0,0 +1,1854 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 325.0, + "eval_steps": 500, + "global_step": 2600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 17550000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2600/training_args.bin b/checkpoint-2600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2650/config.json b/checkpoint-2650/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2650/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2650/generation_config.json b/checkpoint-2650/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2650/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2650/model.safetensors b/checkpoint-2650/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2650/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2650/optimizer.pt b/checkpoint-2650/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..91d8c4379e167de8a00d60533e63ee551fb8e7d7 --- /dev/null +++ b/checkpoint-2650/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9a333d9c87e6659ffb6ffa74ff2d6aecbec4416a5896eee91f7dd0614132261e +size 13823 diff --git a/checkpoint-2650/rng_state.pth b/checkpoint-2650/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..fc7133b7ee34a01715db48df593ec5e7439e5af3 --- /dev/null +++ b/checkpoint-2650/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7713450f304e44c7c5c5aebb57930437deb13fec3b7a8c3c695ab4a1ff58c0a9 +size 14455 diff --git a/checkpoint-2650/scheduler.pt b/checkpoint-2650/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ab0b7dafb95c9daad408b79907587aeb32305a19 --- /dev/null +++ b/checkpoint-2650/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e98adc802225999fb50ca5a8e3d792dde8002e7eb5e81b0efabb488f0651e83 +size 1465 diff --git a/checkpoint-2650/trainer_state.json b/checkpoint-2650/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..8e47161492c8838dfdbf31f78f8ad158e4fbb05e --- /dev/null +++ b/checkpoint-2650/trainer_state.json @@ -0,0 +1,1889 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 331.25, + "eval_steps": 500, + "global_step": 2650, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 17887824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2650/training_args.bin b/checkpoint-2650/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2650/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2700/config.json b/checkpoint-2700/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2700/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2700/generation_config.json b/checkpoint-2700/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2700/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2700/model.safetensors b/checkpoint-2700/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2700/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2700/optimizer.pt b/checkpoint-2700/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f183ea43644e6055cf615ce6fa7c4da5927b8254 --- /dev/null +++ b/checkpoint-2700/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:86aa486d587cd9a20f1ffaf453591fde7601b8c6068733c8b46cc505e73f683e +size 13823 diff --git a/checkpoint-2700/rng_state.pth b/checkpoint-2700/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..502804b9ab5cd434c35b06e257dc5519384c9b28 --- /dev/null +++ b/checkpoint-2700/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f28ba0e62cfbe5e19d7c50cd76caf30eb37e688f5b67bb90c9ac05df28353f88 +size 14455 diff --git a/checkpoint-2700/scheduler.pt b/checkpoint-2700/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..75a8d1c3107de7e4902a8cd154e9bf47fec7075b --- /dev/null +++ b/checkpoint-2700/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:86c3d4c2a6f4acf67570266336692711c8e57d1fd04cd886a793a20d5e6b2dbb +size 1465 diff --git a/checkpoint-2700/trainer_state.json b/checkpoint-2700/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..77a657a0aae955665e40f215fa4e4cd67adc37ec --- /dev/null +++ b/checkpoint-2700/trainer_state.json @@ -0,0 +1,1924 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 337.5, + "eval_steps": 500, + "global_step": 2700, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 18225648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2700/training_args.bin b/checkpoint-2700/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2700/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2750/config.json b/checkpoint-2750/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2750/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2750/generation_config.json b/checkpoint-2750/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2750/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2750/model.safetensors b/checkpoint-2750/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2750/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2750/optimizer.pt b/checkpoint-2750/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..5996954022508da018a080dac5c9dd8c60590b20 --- /dev/null +++ b/checkpoint-2750/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:923a9a7d257c3915b9010b0ecc16cd9ffed382835327d9882519ffc0caf2a4ea +size 13823 diff --git a/checkpoint-2750/rng_state.pth b/checkpoint-2750/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..495aa3b24f0d2620c9ec37fca9c8b8d86b779daf --- /dev/null +++ b/checkpoint-2750/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be8161f9dffa581f3f02bb37148b4eab1832375ef95607e54b668e6cddf91ea9 +size 14455 diff --git a/checkpoint-2750/scheduler.pt b/checkpoint-2750/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..86388361273ff04abc0accee131bbb7485b40ea1 --- /dev/null +++ b/checkpoint-2750/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bd32e8037ef0c3d9a6edd4226d3b0dcf369ef87cf72e99b5538170338eab8f89 +size 1465 diff --git a/checkpoint-2750/trainer_state.json b/checkpoint-2750/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f71867072ce0066aeedaa663c130c733a3ec81ad --- /dev/null +++ b/checkpoint-2750/trainer_state.json @@ -0,0 +1,1959 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 343.75, + "eval_steps": 500, + "global_step": 2750, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 18563472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2750/training_args.bin b/checkpoint-2750/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2750/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2800/config.json b/checkpoint-2800/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2800/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2800/generation_config.json b/checkpoint-2800/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2800/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2800/model.safetensors b/checkpoint-2800/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2800/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2800/optimizer.pt b/checkpoint-2800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..fd4d15e2528da90255c2f94e43b0a2e67943c8ef --- /dev/null +++ b/checkpoint-2800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:99229dbf2b2ce679a5dcdc9e1268d0a59a3a4ab778a964b59b07344f89716ba4 +size 13823 diff --git a/checkpoint-2800/rng_state.pth b/checkpoint-2800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..a5033e7c52bd6bb7b28fdcbb56ce24f96b05e721 --- /dev/null +++ b/checkpoint-2800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:66a9f1ec87d323af7d51c339c1efbd87b932ce9c35f0080072149ec5f434860e +size 14455 diff --git a/checkpoint-2800/scheduler.pt b/checkpoint-2800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..0747e705007b6be221b155d526aacf2b8921ba94 --- /dev/null +++ b/checkpoint-2800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c2e5c7d837c25d78fbea21cb5af15800878be668837070fdfb06f2eab5780dc +size 1465 diff --git a/checkpoint-2800/trainer_state.json b/checkpoint-2800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..91c1119bf02075f9e67a4d9ac34531dad76e997e --- /dev/null +++ b/checkpoint-2800/trainer_state.json @@ -0,0 +1,1994 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 350.0, + "eval_steps": 500, + "global_step": 2800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 18900000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2800/training_args.bin b/checkpoint-2800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2850/config.json b/checkpoint-2850/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2850/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2850/generation_config.json b/checkpoint-2850/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2850/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2850/model.safetensors b/checkpoint-2850/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2850/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2850/optimizer.pt b/checkpoint-2850/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..6b94331864fef15400ec50b7f1f71873f007228c --- /dev/null +++ b/checkpoint-2850/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5d55bd461b41f1a618810cbfa582a5200e90dce23025fa3427a261949221ae43 +size 13823 diff --git a/checkpoint-2850/rng_state.pth b/checkpoint-2850/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..a681ccd82b684c4eb14f004b15c4ac5735a00bf9 --- /dev/null +++ b/checkpoint-2850/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5945b38784cfce8891d5a8f23977daa91b422e02a4775a610e748c1e14880738 +size 14455 diff --git a/checkpoint-2850/scheduler.pt b/checkpoint-2850/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..3a756dfd1743f025ff2b12c3c783b22d8098cf91 --- /dev/null +++ b/checkpoint-2850/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:01ef2beb4979e5cba1257034b044604dce5fd2ed02f9961e96080fdbcda49a15 +size 1465 diff --git a/checkpoint-2850/trainer_state.json b/checkpoint-2850/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..b45805ae6647ce4fb7766e9d5e29a17101bc8391 --- /dev/null +++ b/checkpoint-2850/trainer_state.json @@ -0,0 +1,2029 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 356.25, + "eval_steps": 500, + "global_step": 2850, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 19237824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2850/training_args.bin b/checkpoint-2850/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2850/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2900/config.json b/checkpoint-2900/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2900/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2900/generation_config.json b/checkpoint-2900/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2900/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2900/model.safetensors b/checkpoint-2900/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2900/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2900/optimizer.pt b/checkpoint-2900/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..4c0e326a39f71f36bf457c25d8016278a0a4ebf7 --- /dev/null +++ b/checkpoint-2900/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:856941f52b781f0b54a20570d649198dad60d0a6eb32c42c258f1101a745ec8d +size 13823 diff --git a/checkpoint-2900/rng_state.pth b/checkpoint-2900/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..40e91cf6925a26d6efd53e442584f3a1ca7d6cd0 --- /dev/null +++ b/checkpoint-2900/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:03a0957e39d78b938a4c67a5b5a920bdbc1b5a80765a13f98f0801d9f54d9c7a +size 14455 diff --git a/checkpoint-2900/scheduler.pt b/checkpoint-2900/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..78fd467d8c8ae661cac576b4ad7770d8ca37d849 --- /dev/null +++ b/checkpoint-2900/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:700bd354c113cabbcb450d0a83fee8f146dac816c87e183b5fc67c508b2700f6 +size 1465 diff --git a/checkpoint-2900/trainer_state.json b/checkpoint-2900/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9bcf7093e931673eeaf79fa5fa667a6dd5ecfdc8 --- /dev/null +++ b/checkpoint-2900/trainer_state.json @@ -0,0 +1,2064 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 362.5, + "eval_steps": 500, + "global_step": 2900, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 19575648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2900/training_args.bin b/checkpoint-2900/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2900/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-2950/config.json b/checkpoint-2950/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-2950/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-2950/generation_config.json b/checkpoint-2950/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-2950/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-2950/model.safetensors b/checkpoint-2950/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-2950/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-2950/optimizer.pt b/checkpoint-2950/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f02ae31b5380c08f91186695d46c3b5027d89135 --- /dev/null +++ b/checkpoint-2950/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:71fcbfec2ae28fd7792a42a0451451931badc1ab15c56d12cc4ca3d9ddbc1f32 +size 13823 diff --git a/checkpoint-2950/rng_state.pth b/checkpoint-2950/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..305df0e18c221ced13deef033d9b62b74197ef76 --- /dev/null +++ b/checkpoint-2950/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7775c3f2ee0001294a8ff1365377a62f723b9e53f9fa97429b422ecf8771e71a +size 14455 diff --git a/checkpoint-2950/scheduler.pt b/checkpoint-2950/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ba7d7fd859029946112cecf4c22bd1779fd3eab8 --- /dev/null +++ b/checkpoint-2950/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0fd7619132da56373efebcede4582007955f5195830435f50824f32701e08e00 +size 1465 diff --git a/checkpoint-2950/trainer_state.json b/checkpoint-2950/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..bdbb6aab94e2570e053d3d4ed354795cb06d9632 --- /dev/null +++ b/checkpoint-2950/trainer_state.json @@ -0,0 +1,2099 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 368.75, + "eval_steps": 500, + "global_step": 2950, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 19913472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2950/training_args.bin b/checkpoint-2950/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-2950/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-300/config.json b/checkpoint-300/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-300/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-300/generation_config.json b/checkpoint-300/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-300/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-300/model.safetensors b/checkpoint-300/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-300/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-300/optimizer.pt b/checkpoint-300/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..876e1710671b2a589ed591374e366a88916f6c8b --- /dev/null +++ b/checkpoint-300/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6c1db9936f0f95b8094e9b7b40908635496ad7120c6c507f5f44f5fd1e5f96a4 +size 13823 diff --git a/checkpoint-300/rng_state.pth b/checkpoint-300/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..d65f73d8cfa12436ba48d9a73977f0e85c3bb213 --- /dev/null +++ b/checkpoint-300/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4e3dc2ffd417ca43335dc7e4f9ae2b0c82cd1201b3451b1285b0c2b5c355f483 +size 14455 diff --git a/checkpoint-300/scheduler.pt b/checkpoint-300/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..0f7d71ef20f4f25d8e29747e418f8b06a20db944 --- /dev/null +++ b/checkpoint-300/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:725807a5732d6146241b7c1dfc3f54da5950bc6b53ddfb50eab105e63db2eefc +size 1465 diff --git a/checkpoint-300/trainer_state.json b/checkpoint-300/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..d4a8bd583fda216358dd61590e2d9c148fd6624a --- /dev/null +++ b/checkpoint-300/trainer_state.json @@ -0,0 +1,244 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 37.5, + "eval_steps": 500, + "global_step": 300, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2025648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-300/training_args.bin b/checkpoint-300/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-300/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3000/config.json b/checkpoint-3000/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3000/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3000/generation_config.json b/checkpoint-3000/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3000/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3000/model.safetensors b/checkpoint-3000/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3000/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3000/optimizer.pt b/checkpoint-3000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f4a037369257d8475bb91ac8f97209e48c9884b3 --- /dev/null +++ b/checkpoint-3000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8d57d963e1c420003e0d4d8b863fc59925900f351fe6e6e23958dc8bfa278894 +size 13823 diff --git a/checkpoint-3000/rng_state.pth b/checkpoint-3000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..25889af4b29df16b96dff5def94d65a7651a3099 --- /dev/null +++ b/checkpoint-3000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15af192b94873f688f70f6c12dfb5ce598a0970d2169eb2ed9c5e46a9a9c508d +size 14455 diff --git a/checkpoint-3000/scheduler.pt b/checkpoint-3000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..081689e80e2baac1c08054799c4b3bce51de1a24 --- /dev/null +++ b/checkpoint-3000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:194c522c5552f87fd4e5d03bc93f995fc5022dc9e1a06630e2c41a95d64a8215 +size 1465 diff --git a/checkpoint-3000/trainer_state.json b/checkpoint-3000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6afd1dbfe1db534c1e2051471c6eea33249ab159 --- /dev/null +++ b/checkpoint-3000/trainer_state.json @@ -0,0 +1,2134 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 375.0, + "eval_steps": 500, + "global_step": 3000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 20250000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3000/training_args.bin b/checkpoint-3000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3050/config.json b/checkpoint-3050/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3050/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3050/generation_config.json b/checkpoint-3050/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3050/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3050/model.safetensors b/checkpoint-3050/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3050/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3050/optimizer.pt b/checkpoint-3050/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a66143a93848375fc4f916ba1d62651054b748f2 --- /dev/null +++ b/checkpoint-3050/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5d319fbe36e62bfe8c9c400c58ed57a4b76140529ab4c7e5c829f77370f130b5 +size 13823 diff --git a/checkpoint-3050/rng_state.pth b/checkpoint-3050/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..6407816e49eec7e35167c445dcb33989c1d8aa30 --- /dev/null +++ b/checkpoint-3050/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b75105afadc6bfe3182ffa7c213511b7d3714b6d3f5854c1dfe93b95297ea21f +size 14455 diff --git a/checkpoint-3050/scheduler.pt b/checkpoint-3050/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..6dbf2913108ee3a7a20d540d14202c77d9d6ef00 --- /dev/null +++ b/checkpoint-3050/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bf4159785de8855ea7974e506c148db0f2d1d52b7029b1b8589793e874cebeaf +size 1465 diff --git a/checkpoint-3050/trainer_state.json b/checkpoint-3050/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4e3b0fff66979a5dfff752a5251d2cc60629bf2f --- /dev/null +++ b/checkpoint-3050/trainer_state.json @@ -0,0 +1,2169 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 381.25, + "eval_steps": 500, + "global_step": 3050, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 20587824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3050/training_args.bin b/checkpoint-3050/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3050/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3100/config.json b/checkpoint-3100/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3100/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3100/generation_config.json b/checkpoint-3100/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3100/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3100/model.safetensors b/checkpoint-3100/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3100/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3100/optimizer.pt b/checkpoint-3100/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..274d88ddb7b50daa542f3ccb629fd65086f74646 --- /dev/null +++ b/checkpoint-3100/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dfab3caa29bac9492ca9235cc73c6d55609e7f7ec2c9f4375181c7b51b53bb5b +size 13823 diff --git a/checkpoint-3100/rng_state.pth b/checkpoint-3100/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..2c05c06d7133ad22831e59ffcd36030c5ed938be --- /dev/null +++ b/checkpoint-3100/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eb61926445fe0c36c157bd4dea1616ef82c12867d9d0dde9b47c966e48004e72 +size 14455 diff --git a/checkpoint-3100/scheduler.pt b/checkpoint-3100/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..afcf6ef99eea21d31d9479afbf90d72bdf58fb56 --- /dev/null +++ b/checkpoint-3100/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b45c1232d4e1cb904a61bb4a383f901603a7f7ee05b38bdfc803a7faeb7a7ddf +size 1465 diff --git a/checkpoint-3100/trainer_state.json b/checkpoint-3100/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0a082e0c9e50e84853f82448e850cdc04252a0b3 --- /dev/null +++ b/checkpoint-3100/trainer_state.json @@ -0,0 +1,2204 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 387.5, + "eval_steps": 500, + "global_step": 3100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 20925648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3100/training_args.bin b/checkpoint-3100/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3100/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3150/config.json b/checkpoint-3150/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3150/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3150/generation_config.json b/checkpoint-3150/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3150/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3150/model.safetensors b/checkpoint-3150/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3150/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3150/optimizer.pt b/checkpoint-3150/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..6fa178647d868f1f933ec6ab96098e2a247484d8 --- /dev/null +++ b/checkpoint-3150/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d2c70a125ed19472418931821595300f3f4bcf892c35d895dcc09893cfc52a7f +size 13823 diff --git a/checkpoint-3150/rng_state.pth b/checkpoint-3150/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..c8c7b43a3d210dc19dd233c19c85c16e10fe7acb --- /dev/null +++ b/checkpoint-3150/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ccabd9405e25fa2de7aa3516a44ccb7938bff75845aa18bc40cb952fc86d8429 +size 14455 diff --git a/checkpoint-3150/scheduler.pt b/checkpoint-3150/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..5300e68562c029ae5b9e669dce128e80bb8be5fb --- /dev/null +++ b/checkpoint-3150/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0055a4d04852e4efc9038f44a4cb8429766a99c21400ad1f4f40dc4d56711d69 +size 1465 diff --git a/checkpoint-3150/trainer_state.json b/checkpoint-3150/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..a4a3838ffb7c9592f67ac0e589ebb5a6ed5713cf --- /dev/null +++ b/checkpoint-3150/trainer_state.json @@ -0,0 +1,2239 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 393.75, + "eval_steps": 500, + "global_step": 3150, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 21263472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3150/training_args.bin b/checkpoint-3150/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3150/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3200/config.json b/checkpoint-3200/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3200/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3200/generation_config.json b/checkpoint-3200/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3200/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3200/model.safetensors b/checkpoint-3200/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3200/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3200/optimizer.pt b/checkpoint-3200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..1ff6c7271a3eaa4e58e15de334e4b27d86f5883a --- /dev/null +++ b/checkpoint-3200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d3c91713ccf5051b668880e44c2536abbcd2a13c50aa9ac1e20a37f4b46f52ef +size 13823 diff --git a/checkpoint-3200/rng_state.pth b/checkpoint-3200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..7b9cec8adcfe86aafa2c53474668ff0d06926e2b --- /dev/null +++ b/checkpoint-3200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fd7220073386c39d224e7b9ba85b61d65937a65c9a1b5388255d9b7a51d6d914 +size 14455 diff --git a/checkpoint-3200/scheduler.pt b/checkpoint-3200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..08ab8d63496f908505702be56b0fbb704a82e82f --- /dev/null +++ b/checkpoint-3200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bff4780d6f3bbb902f94fd73483dc3e461c5cf59afcb65b4cfd01d581dd03b9b +size 1465 diff --git a/checkpoint-3200/trainer_state.json b/checkpoint-3200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..a4f6984c0d60b6d9c5e6966ad149d86c71517e45 --- /dev/null +++ b/checkpoint-3200/trainer_state.json @@ -0,0 +1,2274 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 400.0, + "eval_steps": 500, + "global_step": 3200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 21600000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3200/training_args.bin b/checkpoint-3200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3250/config.json b/checkpoint-3250/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3250/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3250/generation_config.json b/checkpoint-3250/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3250/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3250/model.safetensors b/checkpoint-3250/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3250/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3250/optimizer.pt b/checkpoint-3250/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..45327893e9bfaf9f0824422d97cbea9f69a26826 --- /dev/null +++ b/checkpoint-3250/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d314da3c07b1654c8f8067e53e5c0f77f47dfc7acbe65b6944a235a8b2a75347 +size 13823 diff --git a/checkpoint-3250/rng_state.pth b/checkpoint-3250/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..0883a2a3f38e273d52a2abcc9bec2567ebde0514 --- /dev/null +++ b/checkpoint-3250/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bb772eac36b4fa1954a7432d4a50dfbfbbabac1902ff73319248db5785caccd9 +size 14455 diff --git a/checkpoint-3250/scheduler.pt b/checkpoint-3250/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..e01c4aaabf17161e66e655247a7916d562111295 --- /dev/null +++ b/checkpoint-3250/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e1eeef234090f1dbe0c3d4828c925cceda662613ee4c9e374af621b202ebee12 +size 1465 diff --git a/checkpoint-3250/trainer_state.json b/checkpoint-3250/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7f54d59880e9b0a3e0be552cfe531c99e3266c37 --- /dev/null +++ b/checkpoint-3250/trainer_state.json @@ -0,0 +1,2309 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 406.25, + "eval_steps": 500, + "global_step": 3250, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 21937824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3250/training_args.bin b/checkpoint-3250/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3250/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3300/config.json b/checkpoint-3300/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3300/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3300/generation_config.json b/checkpoint-3300/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3300/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3300/model.safetensors b/checkpoint-3300/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3300/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3300/optimizer.pt b/checkpoint-3300/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d8c7aed2af3e5fb89af48cde035df89f581aa24f --- /dev/null +++ b/checkpoint-3300/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:835149d7fdd559f93f8a3f5fa98d141c9eace7cf1a4b4a81e9c1e6d5cf44debc +size 13823 diff --git a/checkpoint-3300/rng_state.pth b/checkpoint-3300/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..de876eded9fec6a1f6920d2585ad10e004c8b4d4 --- /dev/null +++ b/checkpoint-3300/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5a255d8160956b241ee01b5fe887e0289a19a233db92cc7ed0b69af83717bd33 +size 14455 diff --git a/checkpoint-3300/scheduler.pt b/checkpoint-3300/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..2dae21bd1505df3de65b62310460a2de2f7efaa7 --- /dev/null +++ b/checkpoint-3300/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e4746fc39cf992c957fee3079c9924ea7ce158fd426f70aafdf2dc49dddb13ea +size 1465 diff --git a/checkpoint-3300/trainer_state.json b/checkpoint-3300/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ccaf82adf99e77143cb0512f7714cd38968e8653 --- /dev/null +++ b/checkpoint-3300/trainer_state.json @@ -0,0 +1,2344 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 412.5, + "eval_steps": 500, + "global_step": 3300, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 22275648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3300/training_args.bin b/checkpoint-3300/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3300/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3350/config.json b/checkpoint-3350/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3350/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3350/generation_config.json b/checkpoint-3350/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3350/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3350/model.safetensors b/checkpoint-3350/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3350/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3350/optimizer.pt b/checkpoint-3350/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..192a03f2ce630832e574c1e09290a2e62ddd6703 --- /dev/null +++ b/checkpoint-3350/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0ecd63f4497057e1c0523e9a0532af6877c35208212a4d4e0deb4fb3732bfe24 +size 13823 diff --git a/checkpoint-3350/rng_state.pth b/checkpoint-3350/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..1882b1484b656ac05a3333f022430fa40ca39615 --- /dev/null +++ b/checkpoint-3350/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4701fb11ee57e8ddc23f7ef5de5d2c23c640a1f921a7ab5986ec946c9071be06 +size 14455 diff --git a/checkpoint-3350/scheduler.pt b/checkpoint-3350/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..a5e54ac50256e665e4b07d99166348d2d0726b59 --- /dev/null +++ b/checkpoint-3350/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:211f692ae8c1dbacb900492f6c2430ba24387c519e974ca99246e9fe8a418b7a +size 1465 diff --git a/checkpoint-3350/trainer_state.json b/checkpoint-3350/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9d58ddccd2f2bfd9bc7fb81cc3deed9f5672decd --- /dev/null +++ b/checkpoint-3350/trainer_state.json @@ -0,0 +1,2379 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 418.75, + "eval_steps": 500, + "global_step": 3350, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 22613472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3350/training_args.bin b/checkpoint-3350/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3350/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3400/config.json b/checkpoint-3400/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3400/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3400/generation_config.json b/checkpoint-3400/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3400/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3400/model.safetensors b/checkpoint-3400/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3400/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3400/optimizer.pt b/checkpoint-3400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b344c6f6374ebd21e945c291f33fbcb9b8cfc3a9 --- /dev/null +++ b/checkpoint-3400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bba0a9bfc068bb09f47f24d34ba71e630414a15530f316452b6c809a7a36a13a +size 13823 diff --git a/checkpoint-3400/rng_state.pth b/checkpoint-3400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9fb9ebf256fb1dc0fdb6e731437e5f4f300ee189 --- /dev/null +++ b/checkpoint-3400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:00345afa9f5f930dd594aadf387a09bb869a9e398deb26840e9e3484cc58f75f +size 14455 diff --git a/checkpoint-3400/scheduler.pt b/checkpoint-3400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..8fc77fea93fedb53a40917edc031f5d2db07d238 --- /dev/null +++ b/checkpoint-3400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9f680be2a5fe0fe0c37e3d39ccf51bfdca481a5465cd7cbf7b5765cd3fd7ab7d +size 1465 diff --git a/checkpoint-3400/trainer_state.json b/checkpoint-3400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..a26bf017c8b0d3b2c001274419f710bd9360ae78 --- /dev/null +++ b/checkpoint-3400/trainer_state.json @@ -0,0 +1,2414 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 425.0, + "eval_steps": 500, + "global_step": 3400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 22950000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3400/training_args.bin b/checkpoint-3400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3450/config.json b/checkpoint-3450/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3450/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3450/generation_config.json b/checkpoint-3450/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3450/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3450/model.safetensors b/checkpoint-3450/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3450/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3450/optimizer.pt b/checkpoint-3450/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..2ab19806475824b87544046511a7a96d302155c2 --- /dev/null +++ b/checkpoint-3450/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7ab6bc4baad932a8b02f0ede94b52860f269f632d9d02c4a3c17312ffdcbe03b +size 13823 diff --git a/checkpoint-3450/rng_state.pth b/checkpoint-3450/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..529dc8b2b58982d6de8e94dec95ff0040217b231 --- /dev/null +++ b/checkpoint-3450/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:440fc175794ecf08be3e79fd421bc4af9f5e40e1d3a9689dc2533406e1922cb7 +size 14455 diff --git a/checkpoint-3450/scheduler.pt b/checkpoint-3450/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..580f65dc1503a5f41f126bc0ff1153405ffbdbb5 --- /dev/null +++ b/checkpoint-3450/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:37d528eb9ad90c1c7a0459e0524640c6eeb0ae929ffc232bb88380d7cbfb1aff +size 1465 diff --git a/checkpoint-3450/trainer_state.json b/checkpoint-3450/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..8531fd3bb52f027e551896174f02698c227d625d --- /dev/null +++ b/checkpoint-3450/trainer_state.json @@ -0,0 +1,2449 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 431.25, + "eval_steps": 500, + "global_step": 3450, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 23287824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3450/training_args.bin b/checkpoint-3450/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3450/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-350/config.json b/checkpoint-350/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-350/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-350/generation_config.json b/checkpoint-350/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-350/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-350/model.safetensors b/checkpoint-350/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-350/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-350/optimizer.pt b/checkpoint-350/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..8dbc63d90a7050d9221e6af96533e30e43ffc6e8 --- /dev/null +++ b/checkpoint-350/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:458e10ec8a0c9f08deff398ba9691dc1110fd49f897e3d6647f11df2b4b3c339 +size 13823 diff --git a/checkpoint-350/rng_state.pth b/checkpoint-350/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..7718fd684afa4f1e04e6696ef70e4551fadec393 --- /dev/null +++ b/checkpoint-350/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8d32ade6d8781cd244c67a02248dbc3788c711f0fe189fdf7d9954552df38116 +size 14455 diff --git a/checkpoint-350/scheduler.pt b/checkpoint-350/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..00cda1dff091ecff6aad6bb87f8f09b3f6d4a6f6 --- /dev/null +++ b/checkpoint-350/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2e15c1fc3a2d7cdc6c92ad310e972c30db20465a02008a457e94545fcdb3c0c2 +size 1465 diff --git a/checkpoint-350/trainer_state.json b/checkpoint-350/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..54731c952304990487e9d55caed461818abefb43 --- /dev/null +++ b/checkpoint-350/trainer_state.json @@ -0,0 +1,279 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 43.75, + "eval_steps": 500, + "global_step": 350, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2363472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-350/training_args.bin b/checkpoint-350/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-350/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3500/config.json b/checkpoint-3500/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3500/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3500/generation_config.json b/checkpoint-3500/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3500/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3500/model.safetensors b/checkpoint-3500/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3500/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3500/optimizer.pt b/checkpoint-3500/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a052b56dda5db7376fcad450259786f98bf7b7e6 --- /dev/null +++ b/checkpoint-3500/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:034fdbe0a81c21f32b4c73aa2b1470e6ab61aaeb7113f88316495f8df288a634 +size 13823 diff --git a/checkpoint-3500/rng_state.pth b/checkpoint-3500/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..8b4cf50ee875c9adbdf8576541a8bb36fc56aa46 --- /dev/null +++ b/checkpoint-3500/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:480ecc14d8d42ef1a8ae2bef0d99aa0b6c762510a2a51b98fd9fc43fb1a3dd9e +size 14455 diff --git a/checkpoint-3500/scheduler.pt b/checkpoint-3500/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..4546a1f2abce16a90ce34584915a8feb8bcf752c --- /dev/null +++ b/checkpoint-3500/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0dccaf4d8689eaf4849ab7dfd4af291f2fedc2751069a104fe36cbd4f04447f8 +size 1465 diff --git a/checkpoint-3500/trainer_state.json b/checkpoint-3500/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..06f3f62633b983f289310536d0358a384accf3c3 --- /dev/null +++ b/checkpoint-3500/trainer_state.json @@ -0,0 +1,2484 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 437.5, + "eval_steps": 500, + "global_step": 3500, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 23625648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3500/training_args.bin b/checkpoint-3500/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3500/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3550/config.json b/checkpoint-3550/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3550/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3550/generation_config.json b/checkpoint-3550/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3550/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3550/model.safetensors b/checkpoint-3550/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3550/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3550/optimizer.pt b/checkpoint-3550/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..ba90e8648dd3e2e7dc8fc9bcdabb3f06bb1ac600 --- /dev/null +++ b/checkpoint-3550/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e1aada7f796050aa6965217adfb3ff75c6b15f1b7f506ae8e4c8776a5b033e6f +size 13823 diff --git a/checkpoint-3550/rng_state.pth b/checkpoint-3550/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..f24cb95bc2b18fe7ae85cdc274770038dd7ecc65 --- /dev/null +++ b/checkpoint-3550/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce60ac77b02144fc9810dfcfd437173f76b9f1a95248b048a504d78e20ad8ffd +size 14455 diff --git a/checkpoint-3550/scheduler.pt b/checkpoint-3550/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..a2aee9f164db6a076f0f73af5ed1577f9383fc46 --- /dev/null +++ b/checkpoint-3550/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:815a827287754a1efae046b6bb30f4665f5c92fe835b740b5dc8476f514b8c62 +size 1465 diff --git a/checkpoint-3550/trainer_state.json b/checkpoint-3550/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..af1f9fc277aeb275034b895de6389cf554266f46 --- /dev/null +++ b/checkpoint-3550/trainer_state.json @@ -0,0 +1,2519 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 443.75, + "eval_steps": 500, + "global_step": 3550, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 23963472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3550/training_args.bin b/checkpoint-3550/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3550/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3600/config.json b/checkpoint-3600/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3600/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3600/generation_config.json b/checkpoint-3600/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3600/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3600/model.safetensors b/checkpoint-3600/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3600/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3600/optimizer.pt b/checkpoint-3600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a0bbf6c3cfae4175a06937655a4ed7cac772bbae --- /dev/null +++ b/checkpoint-3600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10bf4eaa0dd05c3efcf0ed438a3a2df6c9f25ca5179de4fd3b9fe85d36f7367d +size 13823 diff --git a/checkpoint-3600/rng_state.pth b/checkpoint-3600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..e97041d522b35f010dbb6cb0262a4a580435ad5e --- /dev/null +++ b/checkpoint-3600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f2737f0df9dce4fa79cc1c8a88088d3c56a87603458a18cd914bda257be94442 +size 14455 diff --git a/checkpoint-3600/scheduler.pt b/checkpoint-3600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..27a83a02010eea5b281162eb6dafc2cbf629e18c --- /dev/null +++ b/checkpoint-3600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cc589ad4eb672881b43c1cae060799e79ae5dc33861a5ba31264e38e42726194 +size 1465 diff --git a/checkpoint-3600/trainer_state.json b/checkpoint-3600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0830dce1e4b920baa31f3b3a47bfb7691aa92803 --- /dev/null +++ b/checkpoint-3600/trainer_state.json @@ -0,0 +1,2554 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 450.0, + "eval_steps": 500, + "global_step": 3600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 24300000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3600/training_args.bin b/checkpoint-3600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3650/config.json b/checkpoint-3650/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3650/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3650/generation_config.json b/checkpoint-3650/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3650/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3650/model.safetensors b/checkpoint-3650/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3650/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3650/optimizer.pt b/checkpoint-3650/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..db934d01a19a31a653e428bf3e894741667f5e94 --- /dev/null +++ b/checkpoint-3650/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:696217d943290aad43f876784b17098804d95ae847da10ba6fb38a28c3817fc6 +size 13823 diff --git a/checkpoint-3650/rng_state.pth b/checkpoint-3650/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..482bc20d17c5d1239ff80e3fc1a31a5d7990490a --- /dev/null +++ b/checkpoint-3650/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:048d2f794bbe80a0895f1383762f83bf4de3828f033d6fab76479865a4e42c50 +size 14455 diff --git a/checkpoint-3650/scheduler.pt b/checkpoint-3650/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..fe42f0d7b2ea95d71662338d30663ebe400857bb --- /dev/null +++ b/checkpoint-3650/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cc291d46e51b75e895ac94e084285301f077d40934b3d741af960ecb87f647df +size 1465 diff --git a/checkpoint-3650/trainer_state.json b/checkpoint-3650/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..b840d723f4f9640b1932a13f4a5d9e939d1ae755 --- /dev/null +++ b/checkpoint-3650/trainer_state.json @@ -0,0 +1,2589 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 456.25, + "eval_steps": 500, + "global_step": 3650, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + }, + { + "epoch": 451.25, + "grad_norm": 0.0, + "learning_rate": 0.0009787234042553192, + "loss": 0.0, + "step": 3610 + }, + { + "epoch": 452.5, + "grad_norm": 0.0, + "learning_rate": 0.00095369211514393, + "loss": 0.0, + "step": 3620 + }, + { + "epoch": 453.75, + "grad_norm": 0.0, + "learning_rate": 0.0009286608260325408, + "loss": 0.0, + "step": 3630 + }, + { + "epoch": 455.0, + "grad_norm": 0.0, + "learning_rate": 0.0009036295369211514, + "loss": 0.0, + "step": 3640 + }, + { + "epoch": 456.25, + "grad_norm": 0.0, + "learning_rate": 0.0008785982478097622, + "loss": 0.0, + "step": 3650 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 24637824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3650/training_args.bin b/checkpoint-3650/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3650/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3700/config.json b/checkpoint-3700/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3700/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3700/generation_config.json b/checkpoint-3700/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3700/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3700/model.safetensors b/checkpoint-3700/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3700/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3700/optimizer.pt b/checkpoint-3700/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..15c090642850ebfa2c06abcc7aa827afd7eac50e --- /dev/null +++ b/checkpoint-3700/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c18eeca5d07112e37f2ee47b89abb12ebf4a16ba651f943664b607c456c49026 +size 13823 diff --git a/checkpoint-3700/rng_state.pth b/checkpoint-3700/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..1f0fb8c9c4756a48feca2d3cf78674df46dc90d5 --- /dev/null +++ b/checkpoint-3700/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b767eeb06ecf4c9ddee8531f14dbe3247a89cfcd24112a1c792c0cc442c7a9f9 +size 14455 diff --git a/checkpoint-3700/scheduler.pt b/checkpoint-3700/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..764f3b97bb96102ecea24cb20e4b0cd6e1417d75 --- /dev/null +++ b/checkpoint-3700/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b77d4eaf4902685405124495a16a3fa29e2d93183a762822214212a6f2d48dae +size 1465 diff --git a/checkpoint-3700/trainer_state.json b/checkpoint-3700/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..d993d2047bf0ebf4cc1deba971a6a655462808b6 --- /dev/null +++ b/checkpoint-3700/trainer_state.json @@ -0,0 +1,2624 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 462.5, + "eval_steps": 500, + "global_step": 3700, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + }, + { + "epoch": 451.25, + "grad_norm": 0.0, + "learning_rate": 0.0009787234042553192, + "loss": 0.0, + "step": 3610 + }, + { + "epoch": 452.5, + "grad_norm": 0.0, + "learning_rate": 0.00095369211514393, + "loss": 0.0, + "step": 3620 + }, + { + "epoch": 453.75, + "grad_norm": 0.0, + "learning_rate": 0.0009286608260325408, + "loss": 0.0, + "step": 3630 + }, + { + "epoch": 455.0, + "grad_norm": 0.0, + "learning_rate": 0.0009036295369211514, + "loss": 0.0, + "step": 3640 + }, + { + "epoch": 456.25, + "grad_norm": 0.0, + "learning_rate": 0.0008785982478097622, + "loss": 0.0, + "step": 3650 + }, + { + "epoch": 457.5, + "grad_norm": 0.0, + "learning_rate": 0.000853566958698373, + "loss": 0.0, + "step": 3660 + }, + { + "epoch": 458.75, + "grad_norm": 0.0, + "learning_rate": 0.0008285356695869838, + "loss": 0.0, + "step": 3670 + }, + { + "epoch": 460.0, + "grad_norm": 0.0, + "learning_rate": 0.0008035043804755945, + "loss": 0.0, + "step": 3680 + }, + { + "epoch": 461.25, + "grad_norm": 0.0, + "learning_rate": 0.0007784730913642053, + "loss": 0.0, + "step": 3690 + }, + { + "epoch": 462.5, + "grad_norm": 0.0, + "learning_rate": 0.0007534418022528159, + "loss": 0.0, + "step": 3700 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 24975648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3700/training_args.bin b/checkpoint-3700/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3700/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3750/config.json b/checkpoint-3750/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3750/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3750/generation_config.json b/checkpoint-3750/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3750/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3750/model.safetensors b/checkpoint-3750/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3750/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3750/optimizer.pt b/checkpoint-3750/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..567ae258452ed4e236f410d3e1ad31cf5580e64a --- /dev/null +++ b/checkpoint-3750/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b8de9c7d62c5e03a8e315b08ed1db8a5a9c71c4c3f3548aed8a1db9faa3d0141 +size 13823 diff --git a/checkpoint-3750/rng_state.pth b/checkpoint-3750/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..18f94cbde9bd230e5cb4bdac13d2bc14917b662d --- /dev/null +++ b/checkpoint-3750/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6fac91f6017c143ed40d7e4ae9ed5cf0807222905a3bca96a2a06edbe398ee06 +size 14455 diff --git a/checkpoint-3750/scheduler.pt b/checkpoint-3750/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..80b2de377ae778f00e5fed056baf67b19300d89f --- /dev/null +++ b/checkpoint-3750/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ea92e562ed0ffcd7436101031c4ff57cdeea616761620e77beec30a1d3e0a7d9 +size 1465 diff --git a/checkpoint-3750/trainer_state.json b/checkpoint-3750/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..828d1efad5e8f283cb968a748004abfdfa75b119 --- /dev/null +++ b/checkpoint-3750/trainer_state.json @@ -0,0 +1,2659 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 468.75, + "eval_steps": 500, + "global_step": 3750, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + }, + { + "epoch": 451.25, + "grad_norm": 0.0, + "learning_rate": 0.0009787234042553192, + "loss": 0.0, + "step": 3610 + }, + { + "epoch": 452.5, + "grad_norm": 0.0, + "learning_rate": 0.00095369211514393, + "loss": 0.0, + "step": 3620 + }, + { + "epoch": 453.75, + "grad_norm": 0.0, + "learning_rate": 0.0009286608260325408, + "loss": 0.0, + "step": 3630 + }, + { + "epoch": 455.0, + "grad_norm": 0.0, + "learning_rate": 0.0009036295369211514, + "loss": 0.0, + "step": 3640 + }, + { + "epoch": 456.25, + "grad_norm": 0.0, + "learning_rate": 0.0008785982478097622, + "loss": 0.0, + "step": 3650 + }, + { + "epoch": 457.5, + "grad_norm": 0.0, + "learning_rate": 0.000853566958698373, + "loss": 0.0, + "step": 3660 + }, + { + "epoch": 458.75, + "grad_norm": 0.0, + "learning_rate": 0.0008285356695869838, + "loss": 0.0, + "step": 3670 + }, + { + "epoch": 460.0, + "grad_norm": 0.0, + "learning_rate": 0.0008035043804755945, + "loss": 0.0, + "step": 3680 + }, + { + "epoch": 461.25, + "grad_norm": 0.0, + "learning_rate": 0.0007784730913642053, + "loss": 0.0, + "step": 3690 + }, + { + "epoch": 462.5, + "grad_norm": 0.0, + "learning_rate": 0.0007534418022528159, + "loss": 0.0, + "step": 3700 + }, + { + "epoch": 463.75, + "grad_norm": 0.0, + "learning_rate": 0.0007284105131414268, + "loss": 0.0, + "step": 3710 + }, + { + "epoch": 465.0, + "grad_norm": 0.0, + "learning_rate": 0.0007033792240300375, + "loss": 0.0, + "step": 3720 + }, + { + "epoch": 466.25, + "grad_norm": 0.0, + "learning_rate": 0.0006783479349186483, + "loss": 0.0, + "step": 3730 + }, + { + "epoch": 467.5, + "grad_norm": 0.0, + "learning_rate": 0.0006533166458072592, + "loss": 0.0, + "step": 3740 + }, + { + "epoch": 468.75, + "grad_norm": 0.0, + "learning_rate": 0.0006282853566958699, + "loss": 0.0, + "step": 3750 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 25313472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3750/training_args.bin b/checkpoint-3750/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3750/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3800/config.json b/checkpoint-3800/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3800/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3800/generation_config.json b/checkpoint-3800/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3800/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3800/model.safetensors b/checkpoint-3800/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3800/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3800/optimizer.pt b/checkpoint-3800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b9705c1ffd6764ee1ed6aaa39c29a5448531962e --- /dev/null +++ b/checkpoint-3800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0a3b1d5f01554384e33feab21af8f4fccf46bc89aadbc08f32d78082e10215ea +size 13823 diff --git a/checkpoint-3800/rng_state.pth b/checkpoint-3800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..5a5c84e86d7e4f912381ac823ab1ca78b8d9050f --- /dev/null +++ b/checkpoint-3800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d7a2f2ae7d01a92f1ec84a548785b2be1df4963efb946f812e2c3226afee6a86 +size 14455 diff --git a/checkpoint-3800/scheduler.pt b/checkpoint-3800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..52f5203284bc93d8450a5f8fe46acb46183c82a5 --- /dev/null +++ b/checkpoint-3800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cd160c0053acfdd3fae780147c526a0eb79db71f38ea5d02c2cd012d3e21d95e +size 1465 diff --git a/checkpoint-3800/trainer_state.json b/checkpoint-3800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..291ca5def3af242981cc33f16c403fc9e1c2d34b --- /dev/null +++ b/checkpoint-3800/trainer_state.json @@ -0,0 +1,2694 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 475.0, + "eval_steps": 500, + "global_step": 3800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + }, + { + "epoch": 451.25, + "grad_norm": 0.0, + "learning_rate": 0.0009787234042553192, + "loss": 0.0, + "step": 3610 + }, + { + "epoch": 452.5, + "grad_norm": 0.0, + "learning_rate": 0.00095369211514393, + "loss": 0.0, + "step": 3620 + }, + { + "epoch": 453.75, + "grad_norm": 0.0, + "learning_rate": 0.0009286608260325408, + "loss": 0.0, + "step": 3630 + }, + { + "epoch": 455.0, + "grad_norm": 0.0, + "learning_rate": 0.0009036295369211514, + "loss": 0.0, + "step": 3640 + }, + { + "epoch": 456.25, + "grad_norm": 0.0, + "learning_rate": 0.0008785982478097622, + "loss": 0.0, + "step": 3650 + }, + { + "epoch": 457.5, + "grad_norm": 0.0, + "learning_rate": 0.000853566958698373, + "loss": 0.0, + "step": 3660 + }, + { + "epoch": 458.75, + "grad_norm": 0.0, + "learning_rate": 0.0008285356695869838, + "loss": 0.0, + "step": 3670 + }, + { + "epoch": 460.0, + "grad_norm": 0.0, + "learning_rate": 0.0008035043804755945, + "loss": 0.0, + "step": 3680 + }, + { + "epoch": 461.25, + "grad_norm": 0.0, + "learning_rate": 0.0007784730913642053, + "loss": 0.0, + "step": 3690 + }, + { + "epoch": 462.5, + "grad_norm": 0.0, + "learning_rate": 0.0007534418022528159, + "loss": 0.0, + "step": 3700 + }, + { + "epoch": 463.75, + "grad_norm": 0.0, + "learning_rate": 0.0007284105131414268, + "loss": 0.0, + "step": 3710 + }, + { + "epoch": 465.0, + "grad_norm": 0.0, + "learning_rate": 0.0007033792240300375, + "loss": 0.0, + "step": 3720 + }, + { + "epoch": 466.25, + "grad_norm": 0.0, + "learning_rate": 0.0006783479349186483, + "loss": 0.0, + "step": 3730 + }, + { + "epoch": 467.5, + "grad_norm": 0.0, + "learning_rate": 0.0006533166458072592, + "loss": 0.0, + "step": 3740 + }, + { + "epoch": 468.75, + "grad_norm": 0.0, + "learning_rate": 0.0006282853566958699, + "loss": 0.0, + "step": 3750 + }, + { + "epoch": 470.0, + "grad_norm": 0.0, + "learning_rate": 0.0006032540675844806, + "loss": 0.0, + "step": 3760 + }, + { + "epoch": 471.25, + "grad_norm": 0.0, + "learning_rate": 0.0005782227784730914, + "loss": 0.0, + "step": 3770 + }, + { + "epoch": 472.5, + "grad_norm": 0.0, + "learning_rate": 0.0005531914893617021, + "loss": 0.0, + "step": 3780 + }, + { + "epoch": 473.75, + "grad_norm": 0.0, + "learning_rate": 0.0005281602002503129, + "loss": 0.0, + "step": 3790 + }, + { + "epoch": 475.0, + "grad_norm": 0.0, + "learning_rate": 0.0005031289111389237, + "loss": 0.0, + "step": 3800 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 25650000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3800/training_args.bin b/checkpoint-3800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3850/config.json b/checkpoint-3850/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3850/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3850/generation_config.json b/checkpoint-3850/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3850/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3850/model.safetensors b/checkpoint-3850/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3850/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3850/optimizer.pt b/checkpoint-3850/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..cb907337af3c501da81973fd46579cee86550649 --- /dev/null +++ b/checkpoint-3850/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4bcca6f43be128b249bf5275ceab07dd1a977bd13dc67a586fc4edc7859a2336 +size 13823 diff --git a/checkpoint-3850/rng_state.pth b/checkpoint-3850/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4cca70a3ea1bd9b5ca9462f75e16236486bb8c11 --- /dev/null +++ b/checkpoint-3850/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c75eb3fdd1a6c9d070a19f52726d715d224ebd56a68327df0b3911d85c0a38e6 +size 14455 diff --git a/checkpoint-3850/scheduler.pt b/checkpoint-3850/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..1d6351efafbb9de2b14b0e6df5855d8294523b22 --- /dev/null +++ b/checkpoint-3850/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fa980d7378bd8a84d7d6e96bb8af3454ecc78bd801ec635e8b02647e3e948c2b +size 1465 diff --git a/checkpoint-3850/trainer_state.json b/checkpoint-3850/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..dac50c754712f69afdfd7087f5e02059ce98e4c0 --- /dev/null +++ b/checkpoint-3850/trainer_state.json @@ -0,0 +1,2729 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 481.25, + "eval_steps": 500, + "global_step": 3850, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + }, + { + "epoch": 451.25, + "grad_norm": 0.0, + "learning_rate": 0.0009787234042553192, + "loss": 0.0, + "step": 3610 + }, + { + "epoch": 452.5, + "grad_norm": 0.0, + "learning_rate": 0.00095369211514393, + "loss": 0.0, + "step": 3620 + }, + { + "epoch": 453.75, + "grad_norm": 0.0, + "learning_rate": 0.0009286608260325408, + "loss": 0.0, + "step": 3630 + }, + { + "epoch": 455.0, + "grad_norm": 0.0, + "learning_rate": 0.0009036295369211514, + "loss": 0.0, + "step": 3640 + }, + { + "epoch": 456.25, + "grad_norm": 0.0, + "learning_rate": 0.0008785982478097622, + "loss": 0.0, + "step": 3650 + }, + { + "epoch": 457.5, + "grad_norm": 0.0, + "learning_rate": 0.000853566958698373, + "loss": 0.0, + "step": 3660 + }, + { + "epoch": 458.75, + "grad_norm": 0.0, + "learning_rate": 0.0008285356695869838, + "loss": 0.0, + "step": 3670 + }, + { + "epoch": 460.0, + "grad_norm": 0.0, + "learning_rate": 0.0008035043804755945, + "loss": 0.0, + "step": 3680 + }, + { + "epoch": 461.25, + "grad_norm": 0.0, + "learning_rate": 0.0007784730913642053, + "loss": 0.0, + "step": 3690 + }, + { + "epoch": 462.5, + "grad_norm": 0.0, + "learning_rate": 0.0007534418022528159, + "loss": 0.0, + "step": 3700 + }, + { + "epoch": 463.75, + "grad_norm": 0.0, + "learning_rate": 0.0007284105131414268, + "loss": 0.0, + "step": 3710 + }, + { + "epoch": 465.0, + "grad_norm": 0.0, + "learning_rate": 0.0007033792240300375, + "loss": 0.0, + "step": 3720 + }, + { + "epoch": 466.25, + "grad_norm": 0.0, + "learning_rate": 0.0006783479349186483, + "loss": 0.0, + "step": 3730 + }, + { + "epoch": 467.5, + "grad_norm": 0.0, + "learning_rate": 0.0006533166458072592, + "loss": 0.0, + "step": 3740 + }, + { + "epoch": 468.75, + "grad_norm": 0.0, + "learning_rate": 0.0006282853566958699, + "loss": 0.0, + "step": 3750 + }, + { + "epoch": 470.0, + "grad_norm": 0.0, + "learning_rate": 0.0006032540675844806, + "loss": 0.0, + "step": 3760 + }, + { + "epoch": 471.25, + "grad_norm": 0.0, + "learning_rate": 0.0005782227784730914, + "loss": 0.0, + "step": 3770 + }, + { + "epoch": 472.5, + "grad_norm": 0.0, + "learning_rate": 0.0005531914893617021, + "loss": 0.0, + "step": 3780 + }, + { + "epoch": 473.75, + "grad_norm": 0.0, + "learning_rate": 0.0005281602002503129, + "loss": 0.0, + "step": 3790 + }, + { + "epoch": 475.0, + "grad_norm": 0.0, + "learning_rate": 0.0005031289111389237, + "loss": 0.0, + "step": 3800 + }, + { + "epoch": 476.25, + "grad_norm": 0.0, + "learning_rate": 0.00047809762202753443, + "loss": 0.0, + "step": 3810 + }, + { + "epoch": 477.5, + "grad_norm": 0.0, + "learning_rate": 0.0004530663329161452, + "loss": 0.0, + "step": 3820 + }, + { + "epoch": 478.75, + "grad_norm": 0.0, + "learning_rate": 0.00042803504380475594, + "loss": 0.0, + "step": 3830 + }, + { + "epoch": 480.0, + "grad_norm": 0.0, + "learning_rate": 0.00040300375469336675, + "loss": 0.0, + "step": 3840 + }, + { + "epoch": 481.25, + "grad_norm": 0.0, + "learning_rate": 0.00037797246558197745, + "loss": 0.0, + "step": 3850 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 25987824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3850/training_args.bin b/checkpoint-3850/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3850/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3900/config.json b/checkpoint-3900/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3900/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3900/generation_config.json b/checkpoint-3900/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3900/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3900/model.safetensors b/checkpoint-3900/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3900/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3900/optimizer.pt b/checkpoint-3900/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b30ee2a2ec3398fb279d6b81349873117801e2c2 --- /dev/null +++ b/checkpoint-3900/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e208d9fb2608b588bd1fc00bdf3bda1dae210bde8197880cc182812c4daed449 +size 13823 diff --git a/checkpoint-3900/rng_state.pth b/checkpoint-3900/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..23ec30ae46c84481973de88e20789906e0e515fd --- /dev/null +++ b/checkpoint-3900/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:70efc129331c42567f7989a74b8bed1361bf911b5f35ae50b56a5867b7d63ef2 +size 14455 diff --git a/checkpoint-3900/scheduler.pt b/checkpoint-3900/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..363f24c43cba3b9f3a12f38b50c89565a4cf630f --- /dev/null +++ b/checkpoint-3900/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bf225808344104ffd90ee76c82d47729c6142abfefade838235769cc890afdef +size 1465 diff --git a/checkpoint-3900/trainer_state.json b/checkpoint-3900/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..84e19f1c0b8745e820b00855d1e28c4e2d19a237 --- /dev/null +++ b/checkpoint-3900/trainer_state.json @@ -0,0 +1,2764 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 487.5, + "eval_steps": 500, + "global_step": 3900, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + }, + { + "epoch": 451.25, + "grad_norm": 0.0, + "learning_rate": 0.0009787234042553192, + "loss": 0.0, + "step": 3610 + }, + { + "epoch": 452.5, + "grad_norm": 0.0, + "learning_rate": 0.00095369211514393, + "loss": 0.0, + "step": 3620 + }, + { + "epoch": 453.75, + "grad_norm": 0.0, + "learning_rate": 0.0009286608260325408, + "loss": 0.0, + "step": 3630 + }, + { + "epoch": 455.0, + "grad_norm": 0.0, + "learning_rate": 0.0009036295369211514, + "loss": 0.0, + "step": 3640 + }, + { + "epoch": 456.25, + "grad_norm": 0.0, + "learning_rate": 0.0008785982478097622, + "loss": 0.0, + "step": 3650 + }, + { + "epoch": 457.5, + "grad_norm": 0.0, + "learning_rate": 0.000853566958698373, + "loss": 0.0, + "step": 3660 + }, + { + "epoch": 458.75, + "grad_norm": 0.0, + "learning_rate": 0.0008285356695869838, + "loss": 0.0, + "step": 3670 + }, + { + "epoch": 460.0, + "grad_norm": 0.0, + "learning_rate": 0.0008035043804755945, + "loss": 0.0, + "step": 3680 + }, + { + "epoch": 461.25, + "grad_norm": 0.0, + "learning_rate": 0.0007784730913642053, + "loss": 0.0, + "step": 3690 + }, + { + "epoch": 462.5, + "grad_norm": 0.0, + "learning_rate": 0.0007534418022528159, + "loss": 0.0, + "step": 3700 + }, + { + "epoch": 463.75, + "grad_norm": 0.0, + "learning_rate": 0.0007284105131414268, + "loss": 0.0, + "step": 3710 + }, + { + "epoch": 465.0, + "grad_norm": 0.0, + "learning_rate": 0.0007033792240300375, + "loss": 0.0, + "step": 3720 + }, + { + "epoch": 466.25, + "grad_norm": 0.0, + "learning_rate": 0.0006783479349186483, + "loss": 0.0, + "step": 3730 + }, + { + "epoch": 467.5, + "grad_norm": 0.0, + "learning_rate": 0.0006533166458072592, + "loss": 0.0, + "step": 3740 + }, + { + "epoch": 468.75, + "grad_norm": 0.0, + "learning_rate": 0.0006282853566958699, + "loss": 0.0, + "step": 3750 + }, + { + "epoch": 470.0, + "grad_norm": 0.0, + "learning_rate": 0.0006032540675844806, + "loss": 0.0, + "step": 3760 + }, + { + "epoch": 471.25, + "grad_norm": 0.0, + "learning_rate": 0.0005782227784730914, + "loss": 0.0, + "step": 3770 + }, + { + "epoch": 472.5, + "grad_norm": 0.0, + "learning_rate": 0.0005531914893617021, + "loss": 0.0, + "step": 3780 + }, + { + "epoch": 473.75, + "grad_norm": 0.0, + "learning_rate": 0.0005281602002503129, + "loss": 0.0, + "step": 3790 + }, + { + "epoch": 475.0, + "grad_norm": 0.0, + "learning_rate": 0.0005031289111389237, + "loss": 0.0, + "step": 3800 + }, + { + "epoch": 476.25, + "grad_norm": 0.0, + "learning_rate": 0.00047809762202753443, + "loss": 0.0, + "step": 3810 + }, + { + "epoch": 477.5, + "grad_norm": 0.0, + "learning_rate": 0.0004530663329161452, + "loss": 0.0, + "step": 3820 + }, + { + "epoch": 478.75, + "grad_norm": 0.0, + "learning_rate": 0.00042803504380475594, + "loss": 0.0, + "step": 3830 + }, + { + "epoch": 480.0, + "grad_norm": 0.0, + "learning_rate": 0.00040300375469336675, + "loss": 0.0, + "step": 3840 + }, + { + "epoch": 481.25, + "grad_norm": 0.0, + "learning_rate": 0.00037797246558197745, + "loss": 0.0, + "step": 3850 + }, + { + "epoch": 482.5, + "grad_norm": 0.0, + "learning_rate": 0.00035294117647058826, + "loss": 0.0, + "step": 3860 + }, + { + "epoch": 483.75, + "grad_norm": 0.0, + "learning_rate": 0.000327909887359199, + "loss": 0.0, + "step": 3870 + }, + { + "epoch": 485.0, + "grad_norm": 0.0, + "learning_rate": 0.00030287859824780977, + "loss": 0.0, + "step": 3880 + }, + { + "epoch": 486.25, + "grad_norm": 0.0, + "learning_rate": 0.0002778473091364205, + "loss": 0.0, + "step": 3890 + }, + { + "epoch": 487.5, + "grad_norm": 0.0, + "learning_rate": 0.00025281602002503133, + "loss": 0.0, + "step": 3900 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 26325648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3900/training_args.bin b/checkpoint-3900/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3900/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-3950/config.json b/checkpoint-3950/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-3950/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-3950/generation_config.json b/checkpoint-3950/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-3950/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-3950/model.safetensors b/checkpoint-3950/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-3950/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-3950/optimizer.pt b/checkpoint-3950/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..e55db83fe0d21197e3f351d96af7bf731b93b912 --- /dev/null +++ b/checkpoint-3950/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b8a6bcddf24350953fbda0342a3d98dcbfd5cbc1203e1cd9d75e68839db22ed5 +size 13823 diff --git a/checkpoint-3950/rng_state.pth b/checkpoint-3950/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9ba9cc5497daded28447fdb731d6e950d7b669a7 --- /dev/null +++ b/checkpoint-3950/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:affb1645d3f051398a54c9643541be627632755ab4ca315f1ba67a500877bfbc +size 14455 diff --git a/checkpoint-3950/scheduler.pt b/checkpoint-3950/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..a77aa0b5ef3c35692b816f3f9452f5e3394c5b89 --- /dev/null +++ b/checkpoint-3950/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cc16afcea9423b91f2d1e9c95cdea835ffdfa466ca7e3524b6dc501e82322fd8 +size 1465 diff --git a/checkpoint-3950/trainer_state.json b/checkpoint-3950/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6a22bc0009ae582980cbd795f2f42129265e91f7 --- /dev/null +++ b/checkpoint-3950/trainer_state.json @@ -0,0 +1,2799 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 493.75, + "eval_steps": 500, + "global_step": 3950, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + }, + { + "epoch": 451.25, + "grad_norm": 0.0, + "learning_rate": 0.0009787234042553192, + "loss": 0.0, + "step": 3610 + }, + { + "epoch": 452.5, + "grad_norm": 0.0, + "learning_rate": 0.00095369211514393, + "loss": 0.0, + "step": 3620 + }, + { + "epoch": 453.75, + "grad_norm": 0.0, + "learning_rate": 0.0009286608260325408, + "loss": 0.0, + "step": 3630 + }, + { + "epoch": 455.0, + "grad_norm": 0.0, + "learning_rate": 0.0009036295369211514, + "loss": 0.0, + "step": 3640 + }, + { + "epoch": 456.25, + "grad_norm": 0.0, + "learning_rate": 0.0008785982478097622, + "loss": 0.0, + "step": 3650 + }, + { + "epoch": 457.5, + "grad_norm": 0.0, + "learning_rate": 0.000853566958698373, + "loss": 0.0, + "step": 3660 + }, + { + "epoch": 458.75, + "grad_norm": 0.0, + "learning_rate": 0.0008285356695869838, + "loss": 0.0, + "step": 3670 + }, + { + "epoch": 460.0, + "grad_norm": 0.0, + "learning_rate": 0.0008035043804755945, + "loss": 0.0, + "step": 3680 + }, + { + "epoch": 461.25, + "grad_norm": 0.0, + "learning_rate": 0.0007784730913642053, + "loss": 0.0, + "step": 3690 + }, + { + "epoch": 462.5, + "grad_norm": 0.0, + "learning_rate": 0.0007534418022528159, + "loss": 0.0, + "step": 3700 + }, + { + "epoch": 463.75, + "grad_norm": 0.0, + "learning_rate": 0.0007284105131414268, + "loss": 0.0, + "step": 3710 + }, + { + "epoch": 465.0, + "grad_norm": 0.0, + "learning_rate": 0.0007033792240300375, + "loss": 0.0, + "step": 3720 + }, + { + "epoch": 466.25, + "grad_norm": 0.0, + "learning_rate": 0.0006783479349186483, + "loss": 0.0, + "step": 3730 + }, + { + "epoch": 467.5, + "grad_norm": 0.0, + "learning_rate": 0.0006533166458072592, + "loss": 0.0, + "step": 3740 + }, + { + "epoch": 468.75, + "grad_norm": 0.0, + "learning_rate": 0.0006282853566958699, + "loss": 0.0, + "step": 3750 + }, + { + "epoch": 470.0, + "grad_norm": 0.0, + "learning_rate": 0.0006032540675844806, + "loss": 0.0, + "step": 3760 + }, + { + "epoch": 471.25, + "grad_norm": 0.0, + "learning_rate": 0.0005782227784730914, + "loss": 0.0, + "step": 3770 + }, + { + "epoch": 472.5, + "grad_norm": 0.0, + "learning_rate": 0.0005531914893617021, + "loss": 0.0, + "step": 3780 + }, + { + "epoch": 473.75, + "grad_norm": 0.0, + "learning_rate": 0.0005281602002503129, + "loss": 0.0, + "step": 3790 + }, + { + "epoch": 475.0, + "grad_norm": 0.0, + "learning_rate": 0.0005031289111389237, + "loss": 0.0, + "step": 3800 + }, + { + "epoch": 476.25, + "grad_norm": 0.0, + "learning_rate": 0.00047809762202753443, + "loss": 0.0, + "step": 3810 + }, + { + "epoch": 477.5, + "grad_norm": 0.0, + "learning_rate": 0.0004530663329161452, + "loss": 0.0, + "step": 3820 + }, + { + "epoch": 478.75, + "grad_norm": 0.0, + "learning_rate": 0.00042803504380475594, + "loss": 0.0, + "step": 3830 + }, + { + "epoch": 480.0, + "grad_norm": 0.0, + "learning_rate": 0.00040300375469336675, + "loss": 0.0, + "step": 3840 + }, + { + "epoch": 481.25, + "grad_norm": 0.0, + "learning_rate": 0.00037797246558197745, + "loss": 0.0, + "step": 3850 + }, + { + "epoch": 482.5, + "grad_norm": 0.0, + "learning_rate": 0.00035294117647058826, + "loss": 0.0, + "step": 3860 + }, + { + "epoch": 483.75, + "grad_norm": 0.0, + "learning_rate": 0.000327909887359199, + "loss": 0.0, + "step": 3870 + }, + { + "epoch": 485.0, + "grad_norm": 0.0, + "learning_rate": 0.00030287859824780977, + "loss": 0.0, + "step": 3880 + }, + { + "epoch": 486.25, + "grad_norm": 0.0, + "learning_rate": 0.0002778473091364205, + "loss": 0.0, + "step": 3890 + }, + { + "epoch": 487.5, + "grad_norm": 0.0, + "learning_rate": 0.00025281602002503133, + "loss": 0.0, + "step": 3900 + }, + { + "epoch": 488.75, + "grad_norm": 0.0, + "learning_rate": 0.00022778473091364206, + "loss": 0.0, + "step": 3910 + }, + { + "epoch": 490.0, + "grad_norm": 0.0, + "learning_rate": 0.00020275344180225284, + "loss": 0.0, + "step": 3920 + }, + { + "epoch": 491.25, + "grad_norm": 0.0, + "learning_rate": 0.0001777221526908636, + "loss": 0.0, + "step": 3930 + }, + { + "epoch": 492.5, + "grad_norm": 0.0, + "learning_rate": 0.00015269086357947435, + "loss": 0.0, + "step": 3940 + }, + { + "epoch": 493.75, + "grad_norm": 0.0, + "learning_rate": 0.0001276595744680851, + "loss": 0.0, + "step": 3950 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 26663472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3950/training_args.bin b/checkpoint-3950/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-3950/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-400/config.json b/checkpoint-400/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-400/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-400/generation_config.json b/checkpoint-400/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-400/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-400/model.safetensors b/checkpoint-400/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-400/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-400/optimizer.pt b/checkpoint-400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..5be6c05786305719f677d08c5465c63b31cc9a27 --- /dev/null +++ b/checkpoint-400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:30eb7716a032a51a1eb0f0d759b36db014692d53d9d2e99832a335d2b76ef16c +size 13823 diff --git a/checkpoint-400/rng_state.pth b/checkpoint-400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..b9aac402455a1c12e87140022162b8e0421bb13e --- /dev/null +++ b/checkpoint-400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cc417a91e52df87897c1d609ad03f9b41d1be73043049735c3feb77a9ff53cf8 +size 14455 diff --git a/checkpoint-400/scheduler.pt b/checkpoint-400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ea03e1e2e90492afbcfb4d4fc29df1b3c0f43e9b --- /dev/null +++ b/checkpoint-400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f36712841a7774681459da62a2e3d07eae70458967192affd1c1b09910460655 +size 1465 diff --git a/checkpoint-400/trainer_state.json b/checkpoint-400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e33f0134d0f95ae305a55f5f2cccd5f84a7b60bb --- /dev/null +++ b/checkpoint-400/trainer_state.json @@ -0,0 +1,314 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 50.0, + "eval_steps": 500, + "global_step": 400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2700000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-400/training_args.bin b/checkpoint-400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-4000/config.json b/checkpoint-4000/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-4000/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-4000/generation_config.json b/checkpoint-4000/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-4000/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-4000/model.safetensors b/checkpoint-4000/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-4000/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-4000/optimizer.pt b/checkpoint-4000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..3122c4ae0326837b6aabcc508a264d910e2e0ecd --- /dev/null +++ b/checkpoint-4000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:60720d7ea8b89cbb75460348e94879d40d0fb4e991cdcafd9b3c2cc8bcf2e386 +size 13823 diff --git a/checkpoint-4000/rng_state.pth b/checkpoint-4000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..22fbbea0ca03893870786ebb2dfaff746c1bf982 --- /dev/null +++ b/checkpoint-4000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7d494e738c50c0e43e8de15ff7f9c0ca6721a141dc9b42a6c2f010db42a8d2ee +size 14455 diff --git a/checkpoint-4000/scheduler.pt b/checkpoint-4000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..9c1478218617f26e20b425de64c655b691838066 --- /dev/null +++ b/checkpoint-4000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:60f5ff5a067dbda7237328a703bf0b109d2756d6fcf68921d72e7093e446c131 +size 1465 diff --git a/checkpoint-4000/trainer_state.json b/checkpoint-4000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..338a36b125c8b961d0355c1eb64021cdd24a1e51 --- /dev/null +++ b/checkpoint-4000/trainer_state.json @@ -0,0 +1,2834 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 500.0, + "eval_steps": 500, + "global_step": 4000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + }, + { + "epoch": 120.0, + "grad_norm": 0.0, + "learning_rate": 0.007612015018773467, + "loss": 0.0, + "step": 960 + }, + { + "epoch": 121.25, + "grad_norm": 0.0, + "learning_rate": 0.007586983729662078, + "loss": 0.0, + "step": 970 + }, + { + "epoch": 122.5, + "grad_norm": 0.0, + "learning_rate": 0.007561952440550689, + "loss": 0.0, + "step": 980 + }, + { + "epoch": 123.75, + "grad_norm": 0.0, + "learning_rate": 0.007536921151439299, + "loss": 0.0, + "step": 990 + }, + { + "epoch": 125.0, + "grad_norm": 0.0, + "learning_rate": 0.0075118898623279095, + "loss": 0.0, + "step": 1000 + }, + { + "epoch": 126.25, + "grad_norm": 0.0, + "learning_rate": 0.007486858573216521, + "loss": 0.0, + "step": 1010 + }, + { + "epoch": 127.5, + "grad_norm": 0.0, + "learning_rate": 0.007461827284105132, + "loss": 0.0, + "step": 1020 + }, + { + "epoch": 128.75, + "grad_norm": 0.0, + "learning_rate": 0.007436795994993742, + "loss": 0.0, + "step": 1030 + }, + { + "epoch": 130.0, + "grad_norm": 0.0, + "learning_rate": 0.007411764705882354, + "loss": 0.0, + "step": 1040 + }, + { + "epoch": 131.25, + "grad_norm": 0.0, + "learning_rate": 0.0073867334167709645, + "loss": 0.0, + "step": 1050 + }, + { + "epoch": 132.5, + "grad_norm": 0.0, + "learning_rate": 0.007361702127659574, + "loss": 0.0, + "step": 1060 + }, + { + "epoch": 133.75, + "grad_norm": 0.0, + "learning_rate": 0.007336670838548185, + "loss": 0.0, + "step": 1070 + }, + { + "epoch": 135.0, + "grad_norm": 0.0, + "learning_rate": 0.007311639549436796, + "loss": 0.0, + "step": 1080 + }, + { + "epoch": 136.25, + "grad_norm": 0.0, + "learning_rate": 0.007286608260325407, + "loss": 0.0, + "step": 1090 + }, + { + "epoch": 137.5, + "grad_norm": 0.0, + "learning_rate": 0.007261576971214018, + "loss": 0.0, + "step": 1100 + }, + { + "epoch": 138.75, + "grad_norm": 0.0, + "learning_rate": 0.007236545682102628, + "loss": 0.0, + "step": 1110 + }, + { + "epoch": 140.0, + "grad_norm": 0.0, + "learning_rate": 0.00721151439299124, + "loss": 0.0, + "step": 1120 + }, + { + "epoch": 141.25, + "grad_norm": 0.0, + "learning_rate": 0.00718648310387985, + "loss": 0.0, + "step": 1130 + }, + { + "epoch": 142.5, + "grad_norm": 0.0, + "learning_rate": 0.00716145181476846, + "loss": 0.0, + "step": 1140 + }, + { + "epoch": 143.75, + "grad_norm": 0.0, + "learning_rate": 0.007136420525657071, + "loss": 0.0, + "step": 1150 + }, + { + "epoch": 145.0, + "grad_norm": 0.0, + "learning_rate": 0.007111389236545682, + "loss": 0.0, + "step": 1160 + }, + { + "epoch": 146.25, + "grad_norm": 0.0, + "learning_rate": 0.007086357947434293, + "loss": 0.0, + "step": 1170 + }, + { + "epoch": 147.5, + "grad_norm": 0.0, + "learning_rate": 0.007061326658322904, + "loss": 0.0, + "step": 1180 + }, + { + "epoch": 148.75, + "grad_norm": 0.0, + "learning_rate": 0.007036295369211515, + "loss": 0.0, + "step": 1190 + }, + { + "epoch": 150.0, + "grad_norm": 0.0, + "learning_rate": 0.007011264080100125, + "loss": 0.0, + "step": 1200 + }, + { + "epoch": 151.25, + "grad_norm": 0.0, + "learning_rate": 0.006986232790988736, + "loss": 0.0, + "step": 1210 + }, + { + "epoch": 152.5, + "grad_norm": 0.0, + "learning_rate": 0.006961201501877346, + "loss": 0.0, + "step": 1220 + }, + { + "epoch": 153.75, + "grad_norm": 0.0, + "learning_rate": 0.006936170212765958, + "loss": 0.0, + "step": 1230 + }, + { + "epoch": 155.0, + "grad_norm": 0.0, + "learning_rate": 0.0069111389236545685, + "loss": 0.0, + "step": 1240 + }, + { + "epoch": 156.25, + "grad_norm": 0.0, + "learning_rate": 0.006886107634543179, + "loss": 0.0, + "step": 1250 + }, + { + "epoch": 157.5, + "grad_norm": 0.0, + "learning_rate": 0.006861076345431791, + "loss": 0.0, + "step": 1260 + }, + { + "epoch": 158.75, + "grad_norm": 0.0, + "learning_rate": 0.0068360450563204, + "loss": 0.0, + "step": 1270 + }, + { + "epoch": 160.0, + "grad_norm": 0.0, + "learning_rate": 0.006811013767209011, + "loss": 0.0, + "step": 1280 + }, + { + "epoch": 161.25, + "grad_norm": 0.0, + "learning_rate": 0.006785982478097622, + "loss": 0.0, + "step": 1290 + }, + { + "epoch": 162.5, + "grad_norm": 0.0, + "learning_rate": 0.006760951188986233, + "loss": 0.0, + "step": 1300 + }, + { + "epoch": 163.75, + "grad_norm": 0.0, + "learning_rate": 0.006735919899874844, + "loss": 0.0, + "step": 1310 + }, + { + "epoch": 165.0, + "grad_norm": 0.0, + "learning_rate": 0.0067108886107634545, + "loss": 0.0, + "step": 1320 + }, + { + "epoch": 166.25, + "grad_norm": 0.0, + "learning_rate": 0.006685857321652066, + "loss": 0.0, + "step": 1330 + }, + { + "epoch": 167.5, + "grad_norm": 0.0, + "learning_rate": 0.006660826032540676, + "loss": 0.0, + "step": 1340 + }, + { + "epoch": 168.75, + "grad_norm": 0.0, + "learning_rate": 0.006635794743429286, + "loss": 0.0, + "step": 1350 + }, + { + "epoch": 170.0, + "grad_norm": 0.0, + "learning_rate": 0.006610763454317897, + "loss": 0.0, + "step": 1360 + }, + { + "epoch": 171.25, + "grad_norm": 0.0, + "learning_rate": 0.006585732165206509, + "loss": 0.0, + "step": 1370 + }, + { + "epoch": 172.5, + "grad_norm": 0.0, + "learning_rate": 0.006560700876095119, + "loss": 0.0, + "step": 1380 + }, + { + "epoch": 173.75, + "grad_norm": 0.0, + "learning_rate": 0.00653566958698373, + "loss": 0.0, + "step": 1390 + }, + { + "epoch": 175.0, + "grad_norm": 0.0, + "learning_rate": 0.006510638297872341, + "loss": 0.0, + "step": 1400 + }, + { + "epoch": 176.25, + "grad_norm": 0.0, + "learning_rate": 0.006485607008760951, + "loss": 0.0, + "step": 1410 + }, + { + "epoch": 177.5, + "grad_norm": 0.0, + "learning_rate": 0.006460575719649562, + "loss": 0.0, + "step": 1420 + }, + { + "epoch": 178.75, + "grad_norm": 0.0, + "learning_rate": 0.0064355444305381725, + "loss": 0.0, + "step": 1430 + }, + { + "epoch": 180.0, + "grad_norm": 0.0, + "learning_rate": 0.006410513141426784, + "loss": 0.0, + "step": 1440 + }, + { + "epoch": 181.25, + "grad_norm": 0.0, + "learning_rate": 0.006385481852315395, + "loss": 0.0, + "step": 1450 + }, + { + "epoch": 182.5, + "grad_norm": 0.0, + "learning_rate": 0.006360450563204005, + "loss": 0.0, + "step": 1460 + }, + { + "epoch": 183.75, + "grad_norm": 0.0, + "learning_rate": 0.006335419274092617, + "loss": 0.0, + "step": 1470 + }, + { + "epoch": 185.0, + "grad_norm": 0.0, + "learning_rate": 0.0063103879849812266, + "loss": 0.0, + "step": 1480 + }, + { + "epoch": 186.25, + "grad_norm": 0.0, + "learning_rate": 0.006285356695869837, + "loss": 0.0, + "step": 1490 + }, + { + "epoch": 187.5, + "grad_norm": 0.0, + "learning_rate": 0.006260325406758448, + "loss": 0.0, + "step": 1500 + }, + { + "epoch": 188.75, + "grad_norm": 0.0, + "learning_rate": 0.006235294117647059, + "loss": 0.0, + "step": 1510 + }, + { + "epoch": 190.0, + "grad_norm": 0.0, + "learning_rate": 0.00621026282853567, + "loss": 0.0, + "step": 1520 + }, + { + "epoch": 191.25, + "grad_norm": 0.0, + "learning_rate": 0.006185231539424281, + "loss": 0.0, + "step": 1530 + }, + { + "epoch": 192.5, + "grad_norm": 0.0, + "learning_rate": 0.006160200250312892, + "loss": 0.0, + "step": 1540 + }, + { + "epoch": 193.75, + "grad_norm": 0.0, + "learning_rate": 0.006135168961201502, + "loss": 0.0, + "step": 1550 + }, + { + "epoch": 195.0, + "grad_norm": 0.0, + "learning_rate": 0.006110137672090113, + "loss": 0.0, + "step": 1560 + }, + { + "epoch": 196.25, + "grad_norm": 0.0, + "learning_rate": 0.006085106382978723, + "loss": 0.0, + "step": 1570 + }, + { + "epoch": 197.5, + "grad_norm": 0.0, + "learning_rate": 0.006060075093867335, + "loss": 0.0, + "step": 1580 + }, + { + "epoch": 198.75, + "grad_norm": 0.0, + "learning_rate": 0.006035043804755945, + "loss": 0.0, + "step": 1590 + }, + { + "epoch": 200.0, + "grad_norm": 0.0, + "learning_rate": 0.006010012515644556, + "loss": 0.0, + "step": 1600 + }, + { + "epoch": 201.25, + "grad_norm": 0.0, + "learning_rate": 0.0059849812265331676, + "loss": 0.0, + "step": 1610 + }, + { + "epoch": 202.5, + "grad_norm": 0.0, + "learning_rate": 0.005959949937421777, + "loss": 0.0, + "step": 1620 + }, + { + "epoch": 203.75, + "grad_norm": 0.0, + "learning_rate": 0.005934918648310388, + "loss": 0.0, + "step": 1630 + }, + { + "epoch": 205.0, + "grad_norm": 0.0, + "learning_rate": 0.005909887359198999, + "loss": 0.0, + "step": 1640 + }, + { + "epoch": 206.25, + "grad_norm": 0.0, + "learning_rate": 0.00588485607008761, + "loss": 0.0, + "step": 1650 + }, + { + "epoch": 207.5, + "grad_norm": 0.0, + "learning_rate": 0.005859824780976221, + "loss": 0.0, + "step": 1660 + }, + { + "epoch": 208.75, + "grad_norm": 0.0, + "learning_rate": 0.005834793491864831, + "loss": 0.0, + "step": 1670 + }, + { + "epoch": 210.0, + "grad_norm": 0.0, + "learning_rate": 0.005809762202753442, + "loss": 0.0, + "step": 1680 + }, + { + "epoch": 211.25, + "grad_norm": 0.0, + "learning_rate": 0.005784730913642052, + "loss": 0.0, + "step": 1690 + }, + { + "epoch": 212.5, + "grad_norm": 0.0, + "learning_rate": 0.005759699624530663, + "loss": 0.0, + "step": 1700 + }, + { + "epoch": 213.75, + "grad_norm": 0.0, + "learning_rate": 0.005734668335419274, + "loss": 0.0, + "step": 1710 + }, + { + "epoch": 215.0, + "grad_norm": 0.0, + "learning_rate": 0.005709637046307885, + "loss": 0.0, + "step": 1720 + }, + { + "epoch": 216.25, + "grad_norm": 0.0, + "learning_rate": 0.005684605757196496, + "loss": 0.0, + "step": 1730 + }, + { + "epoch": 217.5, + "grad_norm": 0.0, + "learning_rate": 0.005659574468085107, + "loss": 0.0, + "step": 1740 + }, + { + "epoch": 218.75, + "grad_norm": 0.0, + "learning_rate": 0.005634543178973717, + "loss": 0.0, + "step": 1750 + }, + { + "epoch": 220.0, + "grad_norm": 0.0, + "learning_rate": 0.005609511889862327, + "loss": 0.0, + "step": 1760 + }, + { + "epoch": 221.25, + "grad_norm": 0.0, + "learning_rate": 0.005584480600750939, + "loss": 0.0, + "step": 1770 + }, + { + "epoch": 222.5, + "grad_norm": 0.0, + "learning_rate": 0.005559449311639549, + "loss": 0.0, + "step": 1780 + }, + { + "epoch": 223.75, + "grad_norm": 0.0, + "learning_rate": 0.00553441802252816, + "loss": 0.0, + "step": 1790 + }, + { + "epoch": 225.0, + "grad_norm": 0.0, + "learning_rate": 0.0055093867334167716, + "loss": 0.0, + "step": 1800 + }, + { + "epoch": 226.25, + "grad_norm": 0.0, + "learning_rate": 0.005484355444305382, + "loss": 0.0, + "step": 1810 + }, + { + "epoch": 227.5, + "grad_norm": 0.0, + "learning_rate": 0.005459324155193992, + "loss": 0.0, + "step": 1820 + }, + { + "epoch": 228.75, + "grad_norm": 0.0, + "learning_rate": 0.005434292866082603, + "loss": 0.0, + "step": 1830 + }, + { + "epoch": 230.0, + "grad_norm": 0.0, + "learning_rate": 0.005409261576971214, + "loss": 0.0, + "step": 1840 + }, + { + "epoch": 231.25, + "grad_norm": 0.0, + "learning_rate": 0.005384230287859825, + "loss": 0.0, + "step": 1850 + }, + { + "epoch": 232.5, + "grad_norm": 0.0, + "learning_rate": 0.005359198998748435, + "loss": 0.0, + "step": 1860 + }, + { + "epoch": 233.75, + "grad_norm": 0.0, + "learning_rate": 0.005334167709637047, + "loss": 0.0, + "step": 1870 + }, + { + "epoch": 235.0, + "grad_norm": 0.0, + "learning_rate": 0.005309136420525658, + "loss": 0.0, + "step": 1880 + }, + { + "epoch": 236.25, + "grad_norm": 0.0, + "learning_rate": 0.005284105131414267, + "loss": 0.0, + "step": 1890 + }, + { + "epoch": 237.5, + "grad_norm": 0.0, + "learning_rate": 0.005259073842302878, + "loss": 0.0, + "step": 1900 + }, + { + "epoch": 238.75, + "grad_norm": 0.0, + "learning_rate": 0.0052340425531914895, + "loss": 0.0, + "step": 1910 + }, + { + "epoch": 240.0, + "grad_norm": 0.0, + "learning_rate": 0.0052090112640801, + "loss": 0.0, + "step": 1920 + }, + { + "epoch": 241.25, + "grad_norm": 0.0, + "learning_rate": 0.005183979974968711, + "loss": 0.0, + "step": 1930 + }, + { + "epoch": 242.5, + "grad_norm": 0.0, + "learning_rate": 0.005158948685857322, + "loss": 0.0, + "step": 1940 + }, + { + "epoch": 243.75, + "grad_norm": 0.0, + "learning_rate": 0.005133917396745933, + "loss": 0.0, + "step": 1950 + }, + { + "epoch": 245.0, + "grad_norm": 0.0, + "learning_rate": 0.005108886107634543, + "loss": 0.0, + "step": 1960 + }, + { + "epoch": 246.25, + "grad_norm": 0.0, + "learning_rate": 0.005083854818523153, + "loss": 0.0, + "step": 1970 + }, + { + "epoch": 247.5, + "grad_norm": 0.0, + "learning_rate": 0.005058823529411765, + "loss": 0.0, + "step": 1980 + }, + { + "epoch": 248.75, + "grad_norm": 0.0, + "learning_rate": 0.0050337922403003756, + "loss": 0.0, + "step": 1990 + }, + { + "epoch": 250.0, + "grad_norm": 0.0, + "learning_rate": 0.005008760951188986, + "loss": 0.0, + "step": 2000 + }, + { + "epoch": 251.25, + "grad_norm": 0.0, + "learning_rate": 0.004983729662077598, + "loss": 0.0, + "step": 2010 + }, + { + "epoch": 252.5, + "grad_norm": 0.0, + "learning_rate": 0.0049586983729662075, + "loss": 0.0, + "step": 2020 + }, + { + "epoch": 253.75, + "grad_norm": 0.0, + "learning_rate": 0.004933667083854819, + "loss": 0.0, + "step": 2030 + }, + { + "epoch": 255.0, + "grad_norm": 0.0, + "learning_rate": 0.00490863579474343, + "loss": 0.0, + "step": 2040 + }, + { + "epoch": 256.25, + "grad_norm": 0.0, + "learning_rate": 0.00488360450563204, + "loss": 0.0, + "step": 2050 + }, + { + "epoch": 257.5, + "grad_norm": 0.0, + "learning_rate": 0.004858573216520651, + "loss": 0.0, + "step": 2060 + }, + { + "epoch": 258.75, + "grad_norm": 0.0, + "learning_rate": 0.004833541927409262, + "loss": 0.0, + "step": 2070 + }, + { + "epoch": 260.0, + "grad_norm": 0.0, + "learning_rate": 0.004808510638297872, + "loss": 0.0, + "step": 2080 + }, + { + "epoch": 261.25, + "grad_norm": 0.0, + "learning_rate": 0.004783479349186483, + "loss": 0.0, + "step": 2090 + }, + { + "epoch": 262.5, + "grad_norm": 0.0, + "learning_rate": 0.004758448060075094, + "loss": 0.0, + "step": 2100 + }, + { + "epoch": 263.75, + "grad_norm": 0.0, + "learning_rate": 0.004733416770963705, + "loss": 0.0, + "step": 2110 + }, + { + "epoch": 265.0, + "grad_norm": 0.0, + "learning_rate": 0.004708385481852316, + "loss": 0.0, + "step": 2120 + }, + { + "epoch": 266.25, + "grad_norm": 0.0, + "learning_rate": 0.004683354192740926, + "loss": 0.0, + "step": 2130 + }, + { + "epoch": 267.5, + "grad_norm": 0.0, + "learning_rate": 0.004658322903629537, + "loss": 0.0, + "step": 2140 + }, + { + "epoch": 268.75, + "grad_norm": 0.0, + "learning_rate": 0.004633291614518148, + "loss": 0.0, + "step": 2150 + }, + { + "epoch": 270.0, + "grad_norm": 0.0, + "learning_rate": 0.004608260325406758, + "loss": 0.0, + "step": 2160 + }, + { + "epoch": 271.25, + "grad_norm": 0.0, + "learning_rate": 0.00458322903629537, + "loss": 0.0, + "step": 2170 + }, + { + "epoch": 272.5, + "grad_norm": 0.0, + "learning_rate": 0.00455819774718398, + "loss": 0.0, + "step": 2180 + }, + { + "epoch": 273.75, + "grad_norm": 0.0, + "learning_rate": 0.004533166458072591, + "loss": 0.0, + "step": 2190 + }, + { + "epoch": 275.0, + "grad_norm": 0.0, + "learning_rate": 0.004508135168961202, + "loss": 0.0, + "step": 2200 + }, + { + "epoch": 276.25, + "grad_norm": 0.0, + "learning_rate": 0.004483103879849812, + "loss": 0.0, + "step": 2210 + }, + { + "epoch": 277.5, + "grad_norm": 0.0, + "learning_rate": 0.004458072590738423, + "loss": 0.0, + "step": 2220 + }, + { + "epoch": 278.75, + "grad_norm": 0.0, + "learning_rate": 0.004433041301627034, + "loss": 0.0, + "step": 2230 + }, + { + "epoch": 280.0, + "grad_norm": 0.0, + "learning_rate": 0.004408010012515644, + "loss": 0.0, + "step": 2240 + }, + { + "epoch": 281.25, + "grad_norm": 0.0, + "learning_rate": 0.004382978723404256, + "loss": 0.0, + "step": 2250 + }, + { + "epoch": 282.5, + "grad_norm": 0.0, + "learning_rate": 0.004357947434292866, + "loss": 0.0, + "step": 2260 + }, + { + "epoch": 283.75, + "grad_norm": 0.0, + "learning_rate": 0.004332916145181477, + "loss": 0.0, + "step": 2270 + }, + { + "epoch": 285.0, + "grad_norm": 0.0, + "learning_rate": 0.004307884856070088, + "loss": 0.0, + "step": 2280 + }, + { + "epoch": 286.25, + "grad_norm": 0.0, + "learning_rate": 0.004282853566958698, + "loss": 0.0, + "step": 2290 + }, + { + "epoch": 287.5, + "grad_norm": 0.0, + "learning_rate": 0.004257822277847309, + "loss": 0.0, + "step": 2300 + }, + { + "epoch": 288.75, + "grad_norm": 0.0, + "learning_rate": 0.00423279098873592, + "loss": 0.0, + "step": 2310 + }, + { + "epoch": 290.0, + "grad_norm": 0.0, + "learning_rate": 0.004207759699624531, + "loss": 0.0, + "step": 2320 + }, + { + "epoch": 291.25, + "grad_norm": 0.0, + "learning_rate": 0.004182728410513141, + "loss": 0.0, + "step": 2330 + }, + { + "epoch": 292.5, + "grad_norm": 0.0, + "learning_rate": 0.0041576971214017525, + "loss": 0.0, + "step": 2340 + }, + { + "epoch": 293.75, + "grad_norm": 0.0, + "learning_rate": 0.004132665832290363, + "loss": 0.0, + "step": 2350 + }, + { + "epoch": 295.0, + "grad_norm": 0.0, + "learning_rate": 0.004107634543178974, + "loss": 0.0, + "step": 2360 + }, + { + "epoch": 296.25, + "grad_norm": 0.0, + "learning_rate": 0.004082603254067584, + "loss": 0.0, + "step": 2370 + }, + { + "epoch": 297.5, + "grad_norm": 0.0, + "learning_rate": 0.004057571964956195, + "loss": 0.0, + "step": 2380 + }, + { + "epoch": 298.75, + "grad_norm": 0.0, + "learning_rate": 0.004032540675844807, + "loss": 0.0, + "step": 2390 + }, + { + "epoch": 300.0, + "grad_norm": 0.0, + "learning_rate": 0.004007509386733416, + "loss": 0.0, + "step": 2400 + }, + { + "epoch": 301.25, + "grad_norm": 0.0, + "learning_rate": 0.003982478097622028, + "loss": 0.0, + "step": 2410 + }, + { + "epoch": 302.5, + "grad_norm": 0.0, + "learning_rate": 0.0039574468085106385, + "loss": 0.0, + "step": 2420 + }, + { + "epoch": 303.75, + "grad_norm": 0.0, + "learning_rate": 0.003932415519399249, + "loss": 0.0, + "step": 2430 + }, + { + "epoch": 305.0, + "grad_norm": 0.0, + "learning_rate": 0.00390738423028786, + "loss": 0.0, + "step": 2440 + }, + { + "epoch": 306.25, + "grad_norm": 0.0, + "learning_rate": 0.003882352941176471, + "loss": 0.0, + "step": 2450 + }, + { + "epoch": 307.5, + "grad_norm": 0.0, + "learning_rate": 0.0038573216520650815, + "loss": 0.0, + "step": 2460 + }, + { + "epoch": 308.75, + "grad_norm": 0.0, + "learning_rate": 0.003832290362953692, + "loss": 0.0, + "step": 2470 + }, + { + "epoch": 310.0, + "grad_norm": 0.0, + "learning_rate": 0.003807259073842303, + "loss": 0.0, + "step": 2480 + }, + { + "epoch": 311.25, + "grad_norm": 0.0, + "learning_rate": 0.003782227784730914, + "loss": 0.0, + "step": 2490 + }, + { + "epoch": 312.5, + "grad_norm": 0.0, + "learning_rate": 0.003757196495619524, + "loss": 0.0, + "step": 2500 + }, + { + "epoch": 313.75, + "grad_norm": 0.0, + "learning_rate": 0.003732165206508135, + "loss": 0.0, + "step": 2510 + }, + { + "epoch": 315.0, + "grad_norm": 0.0, + "learning_rate": 0.0037071339173967463, + "loss": 0.0, + "step": 2520 + }, + { + "epoch": 316.25, + "grad_norm": 0.0, + "learning_rate": 0.003682102628285357, + "loss": 0.0, + "step": 2530 + }, + { + "epoch": 317.5, + "grad_norm": 0.0, + "learning_rate": 0.0036570713391739676, + "loss": 0.0, + "step": 2540 + }, + { + "epoch": 318.75, + "grad_norm": 0.0, + "learning_rate": 0.0036320400500625782, + "loss": 0.0, + "step": 2550 + }, + { + "epoch": 320.0, + "grad_norm": 0.0, + "learning_rate": 0.0036070087609511893, + "loss": 0.0, + "step": 2560 + }, + { + "epoch": 321.25, + "grad_norm": 0.0, + "learning_rate": 0.0035819774718397995, + "loss": 0.0, + "step": 2570 + }, + { + "epoch": 322.5, + "grad_norm": 0.0, + "learning_rate": 0.0035569461827284106, + "loss": 0.0, + "step": 2580 + }, + { + "epoch": 323.75, + "grad_norm": 0.0, + "learning_rate": 0.0035319148936170212, + "loss": 0.0, + "step": 2590 + }, + { + "epoch": 325.0, + "grad_norm": 0.0, + "learning_rate": 0.0035068836045056323, + "loss": 0.0, + "step": 2600 + }, + { + "epoch": 326.25, + "grad_norm": 0.0, + "learning_rate": 0.0034818523153942425, + "loss": 0.0, + "step": 2610 + }, + { + "epoch": 327.5, + "grad_norm": 0.0, + "learning_rate": 0.0034568210262828536, + "loss": 0.0, + "step": 2620 + }, + { + "epoch": 328.75, + "grad_norm": 0.0, + "learning_rate": 0.0034317897371714647, + "loss": 0.0, + "step": 2630 + }, + { + "epoch": 330.0, + "grad_norm": 0.0, + "learning_rate": 0.003406758448060075, + "loss": 0.0, + "step": 2640 + }, + { + "epoch": 331.25, + "grad_norm": 0.0, + "learning_rate": 0.003381727158948686, + "loss": 0.0, + "step": 2650 + }, + { + "epoch": 332.5, + "grad_norm": 0.0, + "learning_rate": 0.0033566958698372966, + "loss": 0.0, + "step": 2660 + }, + { + "epoch": 333.75, + "grad_norm": 0.0, + "learning_rate": 0.0033316645807259077, + "loss": 0.0, + "step": 2670 + }, + { + "epoch": 335.0, + "grad_norm": 0.0, + "learning_rate": 0.003306633291614518, + "loss": 0.0, + "step": 2680 + }, + { + "epoch": 336.25, + "grad_norm": 0.0, + "learning_rate": 0.003281602002503129, + "loss": 0.0, + "step": 2690 + }, + { + "epoch": 337.5, + "grad_norm": 0.0, + "learning_rate": 0.00325657071339174, + "loss": 0.0, + "step": 2700 + }, + { + "epoch": 338.75, + "grad_norm": 0.0, + "learning_rate": 0.0032315394242803503, + "loss": 0.0, + "step": 2710 + }, + { + "epoch": 340.0, + "grad_norm": 0.0, + "learning_rate": 0.0032065081351689614, + "loss": 0.0, + "step": 2720 + }, + { + "epoch": 341.25, + "grad_norm": 0.0, + "learning_rate": 0.003181476846057572, + "loss": 0.0, + "step": 2730 + }, + { + "epoch": 342.5, + "grad_norm": 0.0, + "learning_rate": 0.0031564455569461827, + "loss": 0.0, + "step": 2740 + }, + { + "epoch": 343.75, + "grad_norm": 0.0, + "learning_rate": 0.0031314142678347933, + "loss": 0.0, + "step": 2750 + }, + { + "epoch": 345.0, + "grad_norm": 0.0, + "learning_rate": 0.0031063829787234044, + "loss": 0.0, + "step": 2760 + }, + { + "epoch": 346.25, + "grad_norm": 0.0, + "learning_rate": 0.0030813516896120155, + "loss": 0.0, + "step": 2770 + }, + { + "epoch": 347.5, + "grad_norm": 0.0, + "learning_rate": 0.0030563204005006257, + "loss": 0.0, + "step": 2780 + }, + { + "epoch": 348.75, + "grad_norm": 0.0, + "learning_rate": 0.0030312891113892368, + "loss": 0.0, + "step": 2790 + }, + { + "epoch": 350.0, + "grad_norm": 0.0, + "learning_rate": 0.0030062578222778474, + "loss": 0.0, + "step": 2800 + }, + { + "epoch": 351.25, + "grad_norm": 0.0, + "learning_rate": 0.002981226533166458, + "loss": 0.0, + "step": 2810 + }, + { + "epoch": 352.5, + "grad_norm": 0.0, + "learning_rate": 0.0029561952440550687, + "loss": 0.0, + "step": 2820 + }, + { + "epoch": 353.75, + "grad_norm": 0.0, + "learning_rate": 0.0029311639549436798, + "loss": 0.0, + "step": 2830 + }, + { + "epoch": 355.0, + "grad_norm": 0.0, + "learning_rate": 0.0029061326658322904, + "loss": 0.0, + "step": 2840 + }, + { + "epoch": 356.25, + "grad_norm": 0.0, + "learning_rate": 0.002881101376720901, + "loss": 0.0, + "step": 2850 + }, + { + "epoch": 357.5, + "grad_norm": 0.0, + "learning_rate": 0.0028560700876095117, + "loss": 0.0, + "step": 2860 + }, + { + "epoch": 358.75, + "grad_norm": 0.0, + "learning_rate": 0.002831038798498123, + "loss": 0.0, + "step": 2870 + }, + { + "epoch": 360.0, + "grad_norm": 0.0, + "learning_rate": 0.002806007509386733, + "loss": 0.0, + "step": 2880 + }, + { + "epoch": 361.25, + "grad_norm": 0.0, + "learning_rate": 0.002780976220275344, + "loss": 0.0, + "step": 2890 + }, + { + "epoch": 362.5, + "grad_norm": 0.0, + "learning_rate": 0.002755944931163955, + "loss": 0.0, + "step": 2900 + }, + { + "epoch": 363.75, + "grad_norm": 0.0, + "learning_rate": 0.002730913642052566, + "loss": 0.0, + "step": 2910 + }, + { + "epoch": 365.0, + "grad_norm": 0.0, + "learning_rate": 0.0027058823529411765, + "loss": 0.0, + "step": 2920 + }, + { + "epoch": 366.25, + "grad_norm": 0.0, + "learning_rate": 0.002680851063829787, + "loss": 0.0, + "step": 2930 + }, + { + "epoch": 367.5, + "grad_norm": 0.0, + "learning_rate": 0.002655819774718398, + "loss": 0.0, + "step": 2940 + }, + { + "epoch": 368.75, + "grad_norm": 0.0, + "learning_rate": 0.0026307884856070084, + "loss": 0.0, + "step": 2950 + }, + { + "epoch": 370.0, + "grad_norm": 0.0, + "learning_rate": 0.0026057571964956195, + "loss": 0.0, + "step": 2960 + }, + { + "epoch": 371.25, + "grad_norm": 0.0, + "learning_rate": 0.0025807259073842305, + "loss": 0.0, + "step": 2970 + }, + { + "epoch": 372.5, + "grad_norm": 0.0, + "learning_rate": 0.002555694618272841, + "loss": 0.0, + "step": 2980 + }, + { + "epoch": 373.75, + "grad_norm": 0.0, + "learning_rate": 0.002530663329161452, + "loss": 0.0, + "step": 2990 + }, + { + "epoch": 375.0, + "grad_norm": 0.0, + "learning_rate": 0.0025056320400500625, + "loss": 0.0, + "step": 3000 + }, + { + "epoch": 376.25, + "grad_norm": 0.0, + "learning_rate": 0.002480600750938673, + "loss": 0.0, + "step": 3010 + }, + { + "epoch": 377.5, + "grad_norm": 0.0, + "learning_rate": 0.002455569461827284, + "loss": 0.0, + "step": 3020 + }, + { + "epoch": 378.75, + "grad_norm": 0.0, + "learning_rate": 0.002430538172715895, + "loss": 0.0, + "step": 3030 + }, + { + "epoch": 380.0, + "grad_norm": 0.0, + "learning_rate": 0.002405506883604506, + "loss": 0.0, + "step": 3040 + }, + { + "epoch": 381.25, + "grad_norm": 0.0, + "learning_rate": 0.0023804755944931166, + "loss": 0.0, + "step": 3050 + }, + { + "epoch": 382.5, + "grad_norm": 0.0, + "learning_rate": 0.0023554443053817272, + "loss": 0.0, + "step": 3060 + }, + { + "epoch": 383.75, + "grad_norm": 0.0, + "learning_rate": 0.002330413016270338, + "loss": 0.0, + "step": 3070 + }, + { + "epoch": 385.0, + "grad_norm": 0.0, + "learning_rate": 0.0023053817271589485, + "loss": 0.0, + "step": 3080 + }, + { + "epoch": 386.25, + "grad_norm": 0.0, + "learning_rate": 0.0022803504380475596, + "loss": 0.0, + "step": 3090 + }, + { + "epoch": 387.5, + "grad_norm": 0.0, + "learning_rate": 0.0022553191489361702, + "loss": 0.0, + "step": 3100 + }, + { + "epoch": 388.75, + "grad_norm": 0.0, + "learning_rate": 0.0022302878598247813, + "loss": 0.0, + "step": 3110 + }, + { + "epoch": 390.0, + "grad_norm": 0.0, + "learning_rate": 0.002205256570713392, + "loss": 0.0, + "step": 3120 + }, + { + "epoch": 391.25, + "grad_norm": 0.0, + "learning_rate": 0.0021802252816020026, + "loss": 0.0, + "step": 3130 + }, + { + "epoch": 392.5, + "grad_norm": 0.0, + "learning_rate": 0.0021551939924906133, + "loss": 0.0, + "step": 3140 + }, + { + "epoch": 393.75, + "grad_norm": 0.0, + "learning_rate": 0.002130162703379224, + "loss": 0.0, + "step": 3150 + }, + { + "epoch": 395.0, + "grad_norm": 0.0, + "learning_rate": 0.002105131414267835, + "loss": 0.0, + "step": 3160 + }, + { + "epoch": 396.25, + "grad_norm": 0.0, + "learning_rate": 0.0020801001251564456, + "loss": 0.0, + "step": 3170 + }, + { + "epoch": 397.5, + "grad_norm": 0.0, + "learning_rate": 0.0020550688360450563, + "loss": 0.0, + "step": 3180 + }, + { + "epoch": 398.75, + "grad_norm": 0.0, + "learning_rate": 0.002030037546933667, + "loss": 0.0, + "step": 3190 + }, + { + "epoch": 400.0, + "grad_norm": 0.0, + "learning_rate": 0.002005006257822278, + "loss": 0.0, + "step": 3200 + }, + { + "epoch": 401.25, + "grad_norm": 0.0, + "learning_rate": 0.0019799749687108886, + "loss": 0.0, + "step": 3210 + }, + { + "epoch": 402.5, + "grad_norm": 0.0, + "learning_rate": 0.0019549436795994993, + "loss": 0.0, + "step": 3220 + }, + { + "epoch": 403.75, + "grad_norm": 0.0, + "learning_rate": 0.0019299123904881102, + "loss": 0.0, + "step": 3230 + }, + { + "epoch": 405.0, + "grad_norm": 0.0, + "learning_rate": 0.0019048811013767208, + "loss": 0.0, + "step": 3240 + }, + { + "epoch": 406.25, + "grad_norm": 0.0, + "learning_rate": 0.0018798498122653319, + "loss": 0.0, + "step": 3250 + }, + { + "epoch": 407.5, + "grad_norm": 0.0, + "learning_rate": 0.0018548185231539425, + "loss": 0.0, + "step": 3260 + }, + { + "epoch": 408.75, + "grad_norm": 0.0, + "learning_rate": 0.0018297872340425532, + "loss": 0.0, + "step": 3270 + }, + { + "epoch": 410.0, + "grad_norm": 0.0, + "learning_rate": 0.001804755944931164, + "loss": 0.0, + "step": 3280 + }, + { + "epoch": 411.25, + "grad_norm": 0.0, + "learning_rate": 0.0017797246558197747, + "loss": 0.0, + "step": 3290 + }, + { + "epoch": 412.5, + "grad_norm": 0.0, + "learning_rate": 0.0017546933667083855, + "loss": 0.0, + "step": 3300 + }, + { + "epoch": 413.75, + "grad_norm": 0.0, + "learning_rate": 0.0017296620775969962, + "loss": 0.0, + "step": 3310 + }, + { + "epoch": 415.0, + "grad_norm": 0.0, + "learning_rate": 0.001704630788485607, + "loss": 0.0, + "step": 3320 + }, + { + "epoch": 416.25, + "grad_norm": 0.0, + "learning_rate": 0.0016795994993742177, + "loss": 0.0, + "step": 3330 + }, + { + "epoch": 417.5, + "grad_norm": 0.0, + "learning_rate": 0.0016545682102628283, + "loss": 0.0, + "step": 3340 + }, + { + "epoch": 418.75, + "grad_norm": 0.0, + "learning_rate": 0.0016295369211514394, + "loss": 0.0, + "step": 3350 + }, + { + "epoch": 420.0, + "grad_norm": 0.0, + "learning_rate": 0.00160450563204005, + "loss": 0.0, + "step": 3360 + }, + { + "epoch": 421.25, + "grad_norm": 0.0, + "learning_rate": 0.001579474342928661, + "loss": 0.0, + "step": 3370 + }, + { + "epoch": 422.5, + "grad_norm": 0.0, + "learning_rate": 0.0015544430538172716, + "loss": 0.0, + "step": 3380 + }, + { + "epoch": 423.75, + "grad_norm": 0.0, + "learning_rate": 0.0015294117647058824, + "loss": 0.0, + "step": 3390 + }, + { + "epoch": 425.0, + "grad_norm": 0.0, + "learning_rate": 0.001504380475594493, + "loss": 0.0, + "step": 3400 + }, + { + "epoch": 426.25, + "grad_norm": 0.0, + "learning_rate": 0.0014793491864831037, + "loss": 0.0, + "step": 3410 + }, + { + "epoch": 427.5, + "grad_norm": 0.0, + "learning_rate": 0.0014543178973717148, + "loss": 0.0, + "step": 3420 + }, + { + "epoch": 428.75, + "grad_norm": 0.0, + "learning_rate": 0.0014292866082603255, + "loss": 0.0, + "step": 3430 + }, + { + "epoch": 430.0, + "grad_norm": 0.0, + "learning_rate": 0.0014042553191489363, + "loss": 0.0, + "step": 3440 + }, + { + "epoch": 431.25, + "grad_norm": 0.0, + "learning_rate": 0.001379224030037547, + "loss": 0.0, + "step": 3450 + }, + { + "epoch": 432.5, + "grad_norm": 0.0, + "learning_rate": 0.0013541927409261578, + "loss": 0.0, + "step": 3460 + }, + { + "epoch": 433.75, + "grad_norm": 0.0, + "learning_rate": 0.0013291614518147685, + "loss": 0.0, + "step": 3470 + }, + { + "epoch": 435.0, + "grad_norm": 0.0, + "learning_rate": 0.0013041301627033791, + "loss": 0.0, + "step": 3480 + }, + { + "epoch": 436.25, + "grad_norm": 0.0, + "learning_rate": 0.00127909887359199, + "loss": 0.0, + "step": 3490 + }, + { + "epoch": 437.5, + "grad_norm": 0.0, + "learning_rate": 0.0012540675844806006, + "loss": 0.0, + "step": 3500 + }, + { + "epoch": 438.75, + "grad_norm": 0.0, + "learning_rate": 0.0012290362953692115, + "loss": 0.0, + "step": 3510 + }, + { + "epoch": 440.0, + "grad_norm": 0.0, + "learning_rate": 0.0012040050062578223, + "loss": 0.0, + "step": 3520 + }, + { + "epoch": 441.25, + "grad_norm": 0.0, + "learning_rate": 0.001178973717146433, + "loss": 0.0, + "step": 3530 + }, + { + "epoch": 442.5, + "grad_norm": 0.0, + "learning_rate": 0.0011539424280350439, + "loss": 0.0, + "step": 3540 + }, + { + "epoch": 443.75, + "grad_norm": 0.0, + "learning_rate": 0.0011289111389236547, + "loss": 0.0, + "step": 3550 + }, + { + "epoch": 445.0, + "grad_norm": 0.0, + "learning_rate": 0.0011038798498122654, + "loss": 0.0, + "step": 3560 + }, + { + "epoch": 446.25, + "grad_norm": 0.0, + "learning_rate": 0.001078848560700876, + "loss": 0.0, + "step": 3570 + }, + { + "epoch": 447.5, + "grad_norm": 0.0, + "learning_rate": 0.0010538172715894869, + "loss": 0.0, + "step": 3580 + }, + { + "epoch": 448.75, + "grad_norm": 0.0, + "learning_rate": 0.0010287859824780975, + "loss": 0.0, + "step": 3590 + }, + { + "epoch": 450.0, + "grad_norm": 0.0, + "learning_rate": 0.0010037546933667084, + "loss": 0.0, + "step": 3600 + }, + { + "epoch": 451.25, + "grad_norm": 0.0, + "learning_rate": 0.0009787234042553192, + "loss": 0.0, + "step": 3610 + }, + { + "epoch": 452.5, + "grad_norm": 0.0, + "learning_rate": 0.00095369211514393, + "loss": 0.0, + "step": 3620 + }, + { + "epoch": 453.75, + "grad_norm": 0.0, + "learning_rate": 0.0009286608260325408, + "loss": 0.0, + "step": 3630 + }, + { + "epoch": 455.0, + "grad_norm": 0.0, + "learning_rate": 0.0009036295369211514, + "loss": 0.0, + "step": 3640 + }, + { + "epoch": 456.25, + "grad_norm": 0.0, + "learning_rate": 0.0008785982478097622, + "loss": 0.0, + "step": 3650 + }, + { + "epoch": 457.5, + "grad_norm": 0.0, + "learning_rate": 0.000853566958698373, + "loss": 0.0, + "step": 3660 + }, + { + "epoch": 458.75, + "grad_norm": 0.0, + "learning_rate": 0.0008285356695869838, + "loss": 0.0, + "step": 3670 + }, + { + "epoch": 460.0, + "grad_norm": 0.0, + "learning_rate": 0.0008035043804755945, + "loss": 0.0, + "step": 3680 + }, + { + "epoch": 461.25, + "grad_norm": 0.0, + "learning_rate": 0.0007784730913642053, + "loss": 0.0, + "step": 3690 + }, + { + "epoch": 462.5, + "grad_norm": 0.0, + "learning_rate": 0.0007534418022528159, + "loss": 0.0, + "step": 3700 + }, + { + "epoch": 463.75, + "grad_norm": 0.0, + "learning_rate": 0.0007284105131414268, + "loss": 0.0, + "step": 3710 + }, + { + "epoch": 465.0, + "grad_norm": 0.0, + "learning_rate": 0.0007033792240300375, + "loss": 0.0, + "step": 3720 + }, + { + "epoch": 466.25, + "grad_norm": 0.0, + "learning_rate": 0.0006783479349186483, + "loss": 0.0, + "step": 3730 + }, + { + "epoch": 467.5, + "grad_norm": 0.0, + "learning_rate": 0.0006533166458072592, + "loss": 0.0, + "step": 3740 + }, + { + "epoch": 468.75, + "grad_norm": 0.0, + "learning_rate": 0.0006282853566958699, + "loss": 0.0, + "step": 3750 + }, + { + "epoch": 470.0, + "grad_norm": 0.0, + "learning_rate": 0.0006032540675844806, + "loss": 0.0, + "step": 3760 + }, + { + "epoch": 471.25, + "grad_norm": 0.0, + "learning_rate": 0.0005782227784730914, + "loss": 0.0, + "step": 3770 + }, + { + "epoch": 472.5, + "grad_norm": 0.0, + "learning_rate": 0.0005531914893617021, + "loss": 0.0, + "step": 3780 + }, + { + "epoch": 473.75, + "grad_norm": 0.0, + "learning_rate": 0.0005281602002503129, + "loss": 0.0, + "step": 3790 + }, + { + "epoch": 475.0, + "grad_norm": 0.0, + "learning_rate": 0.0005031289111389237, + "loss": 0.0, + "step": 3800 + }, + { + "epoch": 476.25, + "grad_norm": 0.0, + "learning_rate": 0.00047809762202753443, + "loss": 0.0, + "step": 3810 + }, + { + "epoch": 477.5, + "grad_norm": 0.0, + "learning_rate": 0.0004530663329161452, + "loss": 0.0, + "step": 3820 + }, + { + "epoch": 478.75, + "grad_norm": 0.0, + "learning_rate": 0.00042803504380475594, + "loss": 0.0, + "step": 3830 + }, + { + "epoch": 480.0, + "grad_norm": 0.0, + "learning_rate": 0.00040300375469336675, + "loss": 0.0, + "step": 3840 + }, + { + "epoch": 481.25, + "grad_norm": 0.0, + "learning_rate": 0.00037797246558197745, + "loss": 0.0, + "step": 3850 + }, + { + "epoch": 482.5, + "grad_norm": 0.0, + "learning_rate": 0.00035294117647058826, + "loss": 0.0, + "step": 3860 + }, + { + "epoch": 483.75, + "grad_norm": 0.0, + "learning_rate": 0.000327909887359199, + "loss": 0.0, + "step": 3870 + }, + { + "epoch": 485.0, + "grad_norm": 0.0, + "learning_rate": 0.00030287859824780977, + "loss": 0.0, + "step": 3880 + }, + { + "epoch": 486.25, + "grad_norm": 0.0, + "learning_rate": 0.0002778473091364205, + "loss": 0.0, + "step": 3890 + }, + { + "epoch": 487.5, + "grad_norm": 0.0, + "learning_rate": 0.00025281602002503133, + "loss": 0.0, + "step": 3900 + }, + { + "epoch": 488.75, + "grad_norm": 0.0, + "learning_rate": 0.00022778473091364206, + "loss": 0.0, + "step": 3910 + }, + { + "epoch": 490.0, + "grad_norm": 0.0, + "learning_rate": 0.00020275344180225284, + "loss": 0.0, + "step": 3920 + }, + { + "epoch": 491.25, + "grad_norm": 0.0, + "learning_rate": 0.0001777221526908636, + "loss": 0.0, + "step": 3930 + }, + { + "epoch": 492.5, + "grad_norm": 0.0, + "learning_rate": 0.00015269086357947435, + "loss": 0.0, + "step": 3940 + }, + { + "epoch": 493.75, + "grad_norm": 0.0, + "learning_rate": 0.0001276595744680851, + "loss": 0.0, + "step": 3950 + }, + { + "epoch": 495.0, + "grad_norm": 0.0, + "learning_rate": 0.00010262828535669587, + "loss": 0.0, + "step": 3960 + }, + { + "epoch": 496.25, + "grad_norm": 0.0, + "learning_rate": 7.759699624530664e-05, + "loss": 0.0, + "step": 3970 + }, + { + "epoch": 497.5, + "grad_norm": 0.0, + "learning_rate": 5.2565707133917396e-05, + "loss": 0.0, + "step": 3980 + }, + { + "epoch": 498.75, + "grad_norm": 0.0, + "learning_rate": 2.753441802252816e-05, + "loss": 0.0, + "step": 3990 + }, + { + "epoch": 500.0, + "grad_norm": 0.0, + "learning_rate": 2.5031289111389237e-06, + "loss": 0.0, + "step": 4000 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 27000000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-4000/training_args.bin b/checkpoint-4000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-4000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-450/config.json b/checkpoint-450/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-450/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-450/generation_config.json b/checkpoint-450/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-450/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-450/model.safetensors b/checkpoint-450/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-450/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-450/optimizer.pt b/checkpoint-450/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..ec826429e036b284778101a20b1eed45b3e9e086 --- /dev/null +++ b/checkpoint-450/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:64c9c03a4456f42cffd4479dc4725ab2b0dc1d383c94b4501aeba9e635912e4d +size 13823 diff --git a/checkpoint-450/rng_state.pth b/checkpoint-450/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..2c0816f881f317973ee1697c2801a3dc10dd866b --- /dev/null +++ b/checkpoint-450/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a1a39b2c7ba948c919cf511140589fbb763d75febc6b7d7740d7c52282b9e203 +size 14455 diff --git a/checkpoint-450/scheduler.pt b/checkpoint-450/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..dee0d736347d21aefc96a66354dc7b09d6917a05 --- /dev/null +++ b/checkpoint-450/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9e2cc9feac4812fe7f0cd4a7895682445178d54e54da7a4c1344cb7f50075f64 +size 1465 diff --git a/checkpoint-450/trainer_state.json b/checkpoint-450/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6e39859151ade5df74716ef6961e8dd41a5f3fb7 --- /dev/null +++ b/checkpoint-450/trainer_state.json @@ -0,0 +1,349 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 56.25, + "eval_steps": 500, + "global_step": 450, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3037824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-450/training_args.bin b/checkpoint-450/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-450/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-50/config.json b/checkpoint-50/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-50/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-50/generation_config.json b/checkpoint-50/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-50/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-50/model.safetensors b/checkpoint-50/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-50/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-50/optimizer.pt b/checkpoint-50/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..263343cfec451f3801fcaf10c53d3128cc696284 --- /dev/null +++ b/checkpoint-50/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ee82169c5dbf2c6dde5dc6fdff8ee3a164243f2723913a338f48bdb94a614313 +size 13823 diff --git a/checkpoint-50/rng_state.pth b/checkpoint-50/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..d0bc61ca06cc2c1a21a3c783405a546bc9b4b96a --- /dev/null +++ b/checkpoint-50/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:799af21e8c98b5b33efb40e1806882d89f329486619fd8f7f6c267e08bc7954e +size 14455 diff --git a/checkpoint-50/scheduler.pt b/checkpoint-50/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..5a99c2bd7ebd2533cb9d4b68eba6e5bfd83efd4e --- /dev/null +++ b/checkpoint-50/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:74e832a03a9823d252bd8741a2c8b348070cf992dd64c07736391965f2ce0ad1 +size 1465 diff --git a/checkpoint-50/trainer_state.json b/checkpoint-50/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..abb1e0921f2291283b6c88bcb7362c78b4a16ffe --- /dev/null +++ b/checkpoint-50/trainer_state.json @@ -0,0 +1,69 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 6.25, + "eval_steps": 500, + "global_step": 50, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 337824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-50/training_args.bin b/checkpoint-50/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-50/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-500/config.json b/checkpoint-500/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-500/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-500/generation_config.json b/checkpoint-500/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-500/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-500/model.safetensors b/checkpoint-500/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-500/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-500/optimizer.pt b/checkpoint-500/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..032ded771acb81e3acb5e2a708c3e1f0c79e847a --- /dev/null +++ b/checkpoint-500/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e698b4a9f1ba7a61dd47fd0cd7b979a3764b4bb8ac86da5f8bde9291619e0301 +size 13823 diff --git a/checkpoint-500/rng_state.pth b/checkpoint-500/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bf2bc3f0aaaea0ec87fbdcd44bc5f8d3a52cccf9 --- /dev/null +++ b/checkpoint-500/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:03900c236c7794bc0687441d4bebf000599935b0a4712879e57581ae5ba16b25 +size 14455 diff --git a/checkpoint-500/scheduler.pt b/checkpoint-500/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..52e83f46889c738207c34cc595f613eaa907e669 --- /dev/null +++ b/checkpoint-500/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2461134802868888c34f2e03d39a80c79021259bdac977b9668eacd5c9124b89 +size 1465 diff --git a/checkpoint-500/trainer_state.json b/checkpoint-500/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..24b196f76fc5d6ac26a4662c539e487832ffd379 --- /dev/null +++ b/checkpoint-500/trainer_state.json @@ -0,0 +1,384 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 62.5, + "eval_steps": 500, + "global_step": 500, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3375648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-500/training_args.bin b/checkpoint-500/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-500/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-550/config.json b/checkpoint-550/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-550/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-550/generation_config.json b/checkpoint-550/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-550/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-550/model.safetensors b/checkpoint-550/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-550/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-550/optimizer.pt b/checkpoint-550/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..4c11b78aa674c809c8996c37570030d9fee21f1b --- /dev/null +++ b/checkpoint-550/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5677c697e5274a6870620a64b39d15397205fe7d47618d525de79fc38abfbe94 +size 13823 diff --git a/checkpoint-550/rng_state.pth b/checkpoint-550/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4ba514a56c627cc8fb8cc007bbcd509b7bef03b8 --- /dev/null +++ b/checkpoint-550/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8661a2a6fbc6b2c3c1692b0d42731d1ba2b65839ba891f23dcb54dbcafd155d0 +size 14455 diff --git a/checkpoint-550/scheduler.pt b/checkpoint-550/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..7ea0c05e7037c2c43a605fc423c9ff11529b8863 --- /dev/null +++ b/checkpoint-550/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92955ac0979c34678c11b6c41ad9da9d981f2580a753dd80906e07ea343fbdad +size 1465 diff --git a/checkpoint-550/trainer_state.json b/checkpoint-550/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7306a348fa6ca1446b10cc5180d84b36bfc6ee38 --- /dev/null +++ b/checkpoint-550/trainer_state.json @@ -0,0 +1,419 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 68.75, + "eval_steps": 500, + "global_step": 550, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3713472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-550/training_args.bin b/checkpoint-550/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-550/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-600/config.json b/checkpoint-600/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-600/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-600/generation_config.json b/checkpoint-600/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-600/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-600/model.safetensors b/checkpoint-600/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-600/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-600/optimizer.pt b/checkpoint-600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..aac8a3a8ff8b92a5786a241595ef3628be9ae2e5 --- /dev/null +++ b/checkpoint-600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f9aa377620bfcdc96010aaba1757476f61a7bc74b508c5e256c6d4147173646c +size 13823 diff --git a/checkpoint-600/rng_state.pth b/checkpoint-600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..1317304f0b1064d93ee52a59dabfa3cfefe38a3b --- /dev/null +++ b/checkpoint-600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:faf5dcedef23ff9589b3f1c7e8a0a7a39a88cc9c6cc13e5d31cc40d8dd277106 +size 14455 diff --git a/checkpoint-600/scheduler.pt b/checkpoint-600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..227fd9020adbaea14e7503b625a89ad5b0e349a2 --- /dev/null +++ b/checkpoint-600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3e13112a111c5e916993f702855c04f2c4e8621f063cc1de3a06b4d9637f304c +size 1465 diff --git a/checkpoint-600/trainer_state.json b/checkpoint-600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..d70b4cf48aa24a25dc0f835d2265c31f34b7e74e --- /dev/null +++ b/checkpoint-600/trainer_state.json @@ -0,0 +1,454 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 75.0, + "eval_steps": 500, + "global_step": 600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4050000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-600/training_args.bin b/checkpoint-600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-650/config.json b/checkpoint-650/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-650/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-650/generation_config.json b/checkpoint-650/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-650/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-650/model.safetensors b/checkpoint-650/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-650/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-650/optimizer.pt b/checkpoint-650/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..0bb1e73b7f7583ba905e2836b276a55828279776 --- /dev/null +++ b/checkpoint-650/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b8cb974964bb9509465775d5dd67b6f5291b06f03c3f0cf98eb2465364f86ef9 +size 13823 diff --git a/checkpoint-650/rng_state.pth b/checkpoint-650/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..1c95705c48969c1ec221d0701dca8a4965547c6f --- /dev/null +++ b/checkpoint-650/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8da113205091b74335dae201a4d99876ca5fa7a74f35ca366721a829fe7a90f6 +size 14455 diff --git a/checkpoint-650/scheduler.pt b/checkpoint-650/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ce3693c6511f8c7e59149772151b748d04c70af1 --- /dev/null +++ b/checkpoint-650/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d276b6042038289b122d90fabc5d3b1ec27652567b7574c0804d505f97f357d5 +size 1465 diff --git a/checkpoint-650/trainer_state.json b/checkpoint-650/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e4ac34549cf10206dc4f36bbf0942f1a6f2e08d5 --- /dev/null +++ b/checkpoint-650/trainer_state.json @@ -0,0 +1,489 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 81.25, + "eval_steps": 500, + "global_step": 650, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4387824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-650/training_args.bin b/checkpoint-650/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-650/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-700/config.json b/checkpoint-700/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-700/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-700/generation_config.json b/checkpoint-700/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-700/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-700/model.safetensors b/checkpoint-700/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-700/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-700/optimizer.pt b/checkpoint-700/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..3a7ee99476b18a817298afe8c2ad18fc82dc0813 --- /dev/null +++ b/checkpoint-700/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dce76e4591c42e139d583de9308af0a68cb951345d8f7c2a773615a667b074f5 +size 13823 diff --git a/checkpoint-700/rng_state.pth b/checkpoint-700/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bc258f2be2c60fa3760111e381fa190269574ee7 --- /dev/null +++ b/checkpoint-700/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8fd55b291025768703858dbad7135c3d30b95aca5fb1facb3fbcfa8bea7d4aed +size 14455 diff --git a/checkpoint-700/scheduler.pt b/checkpoint-700/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..358411ce5de2e89b8c5fba56132b3011bfdf2c3d --- /dev/null +++ b/checkpoint-700/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:def379bca2f078bebe8befda2d5b439a8f558cfb449fff6776dea41a469d5d24 +size 1465 diff --git a/checkpoint-700/trainer_state.json b/checkpoint-700/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..32c95a586aff388057f0daad9c6f3ab9295aee50 --- /dev/null +++ b/checkpoint-700/trainer_state.json @@ -0,0 +1,524 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 87.5, + "eval_steps": 500, + "global_step": 700, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4725648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-700/training_args.bin b/checkpoint-700/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-700/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-750/config.json b/checkpoint-750/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-750/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-750/generation_config.json b/checkpoint-750/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-750/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-750/model.safetensors b/checkpoint-750/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-750/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-750/optimizer.pt b/checkpoint-750/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..fc5022a5d32fa7b93174380d05d3a892988609be --- /dev/null +++ b/checkpoint-750/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7425247425d03050d4aef2cf20ca6a0d5dfee8504ac9815c88a8af470ab3fb0f +size 13823 diff --git a/checkpoint-750/rng_state.pth b/checkpoint-750/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..a506361c12019ec83bb73a7a236d9a9d48830fe1 --- /dev/null +++ b/checkpoint-750/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e1f43dca5b3841b47ae65a958c5577b390cc73e22ddc783e00335997202850c4 +size 14455 diff --git a/checkpoint-750/scheduler.pt b/checkpoint-750/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..3a5ff6eec7aeaa5d119ac456ef6fb2840bc1dbf2 --- /dev/null +++ b/checkpoint-750/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:83da14ca3358dc3696b4528101acd88e548fccace985059f23be641c03a521b6 +size 1465 diff --git a/checkpoint-750/trainer_state.json b/checkpoint-750/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1750bbcc0a353e3850234d148b92f56bc3ccecde --- /dev/null +++ b/checkpoint-750/trainer_state.json @@ -0,0 +1,559 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 93.75, + "eval_steps": 500, + "global_step": 750, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5063472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-750/training_args.bin b/checkpoint-750/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-750/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-800/config.json b/checkpoint-800/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-800/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-800/generation_config.json b/checkpoint-800/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-800/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-800/model.safetensors b/checkpoint-800/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-800/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-800/optimizer.pt b/checkpoint-800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..298725d44510d804459c1826d78419b8363973a0 --- /dev/null +++ b/checkpoint-800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d37af320acdf186f2859cb9baffdb5ab24c44d624b8010d5680c07cf37b84793 +size 13823 diff --git a/checkpoint-800/rng_state.pth b/checkpoint-800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4e05f8683988632a7c25c1b596130bea09cf1fe5 --- /dev/null +++ b/checkpoint-800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:05d9251e4283d8763f8ab7a5cfc8c298ac585c4fbdf905647f3083c10d1cd929 +size 14455 diff --git a/checkpoint-800/scheduler.pt b/checkpoint-800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..867dbeabc0ec88f557c4cf97dc7168017a8411d9 --- /dev/null +++ b/checkpoint-800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ed9aa5ffa95bf9dbe9ea8958872b72a5b81d0fe6902edba907d31f9a9fb87cfa +size 1465 diff --git a/checkpoint-800/trainer_state.json b/checkpoint-800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..5526c5a58bb1f2c7bdae216b6c23caf45109d46c --- /dev/null +++ b/checkpoint-800/trainer_state.json @@ -0,0 +1,594 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 100.0, + "eval_steps": 500, + "global_step": 800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5400000.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-800/training_args.bin b/checkpoint-800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-850/config.json b/checkpoint-850/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-850/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-850/generation_config.json b/checkpoint-850/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-850/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-850/model.safetensors b/checkpoint-850/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-850/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-850/optimizer.pt b/checkpoint-850/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b03292377015d15a02a36aa6f246df94f80a5da5 --- /dev/null +++ b/checkpoint-850/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d278f84a3e346e8c11dfe7cd0d33083519fc760f4eade37bf80e813e11a29813 +size 13823 diff --git a/checkpoint-850/rng_state.pth b/checkpoint-850/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..96363edb90034871a718e8b4098a6c0c350c3a88 --- /dev/null +++ b/checkpoint-850/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5acc4699e01f5cc3b20cdf0cd8df5eb4abd83fafe2390c064d90eb1e8ef01803 +size 14455 diff --git a/checkpoint-850/scheduler.pt b/checkpoint-850/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..1c1db32689222e118494ece9c29ffd32b7bb6c68 --- /dev/null +++ b/checkpoint-850/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1cb4f176a0cee8b12d9b3d33c073ff8d6f05401e97cc16924a2a884e60ef1482 +size 1465 diff --git a/checkpoint-850/trainer_state.json b/checkpoint-850/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ed257aa98f2e5693976d892f195f9932a87fcfdf --- /dev/null +++ b/checkpoint-850/trainer_state.json @@ -0,0 +1,629 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 106.25, + "eval_steps": 500, + "global_step": 850, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5737824.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-850/training_args.bin b/checkpoint-850/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-850/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-900/config.json b/checkpoint-900/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-900/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-900/generation_config.json b/checkpoint-900/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-900/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-900/model.safetensors b/checkpoint-900/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-900/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-900/optimizer.pt b/checkpoint-900/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..821bf325502b94deb1c5d8ed9da18e4ed49eab70 --- /dev/null +++ b/checkpoint-900/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d7aab34e75d09500c0ff86edefac127ed6cd358cba018425ad29df9a4ca8afb3 +size 13823 diff --git a/checkpoint-900/rng_state.pth b/checkpoint-900/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..422ceac95b71c952b67d1ed22b7814afdb170450 --- /dev/null +++ b/checkpoint-900/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a8aacb968579f13b30d632401908a5b469f011b5e61c26e3299f08aeecb0946b +size 14455 diff --git a/checkpoint-900/scheduler.pt b/checkpoint-900/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..0f72a9b10ea9fcbdbd5d6fec92f4ddef991d8cab --- /dev/null +++ b/checkpoint-900/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3e64cb611f588ed9f8b4dabda142624400a6d22c3baada303db37471726aa710 +size 1465 diff --git a/checkpoint-900/trainer_state.json b/checkpoint-900/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4abec13f42a5a48d264697051a153b22dcac9dd4 --- /dev/null +++ b/checkpoint-900/trainer_state.json @@ -0,0 +1,664 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 112.5, + "eval_steps": 500, + "global_step": 900, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6075648.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-900/training_args.bin b/checkpoint-900/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-900/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/checkpoint-950/config.json b/checkpoint-950/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/checkpoint-950/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/checkpoint-950/generation_config.json b/checkpoint-950/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/checkpoint-950/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/checkpoint-950/model.safetensors b/checkpoint-950/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/checkpoint-950/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/checkpoint-950/optimizer.pt b/checkpoint-950/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b6e2d65be64c6659551ec7cfa9948ff41a8b80ed --- /dev/null +++ b/checkpoint-950/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c11d5809df000aff1490d4d0521e7bdeb7f311fbd6910601fac8cb77922d7052 +size 13823 diff --git a/checkpoint-950/rng_state.pth b/checkpoint-950/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4e560bda356376ab5449d99faa89734e7223868f --- /dev/null +++ b/checkpoint-950/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce6bbb110df837201a2cda492016077e75bdd691144d8ed3c6983854423d6aea +size 14455 diff --git a/checkpoint-950/scheduler.pt b/checkpoint-950/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..832deb4879b9bff2cbb03f7704a7012c1ed936b5 --- /dev/null +++ b/checkpoint-950/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fed33ce740ce870cb93e90ab37d7c5a4517b68b541c90e76dab5eafcf041826 +size 1465 diff --git a/checkpoint-950/trainer_state.json b/checkpoint-950/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..98b744376c05a1538cad189c44f76af5f4115e36 --- /dev/null +++ b/checkpoint-950/trainer_state.json @@ -0,0 +1,699 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 118.75, + "eval_steps": 500, + "global_step": 950, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 1.25, + "grad_norm": 0.0, + "learning_rate": 0.009989987484355445, + "loss": 0.0, + "step": 10 + }, + { + "epoch": 2.5, + "grad_norm": 0.0, + "learning_rate": 0.009964956195244054, + "loss": 0.0, + "step": 20 + }, + { + "epoch": 3.75, + "grad_norm": 0.0, + "learning_rate": 0.009939924906132666, + "loss": 0.0, + "step": 30 + }, + { + "epoch": 5.0, + "grad_norm": 0.0, + "learning_rate": 0.009914893617021277, + "loss": 0.0, + "step": 40 + }, + { + "epoch": 6.25, + "grad_norm": 0.0, + "learning_rate": 0.009889862327909888, + "loss": 0.0, + "step": 50 + }, + { + "epoch": 7.5, + "grad_norm": 0.0, + "learning_rate": 0.009864831038798498, + "loss": 0.0, + "step": 60 + }, + { + "epoch": 8.75, + "grad_norm": 0.0, + "learning_rate": 0.009839799749687109, + "loss": 0.0, + "step": 70 + }, + { + "epoch": 10.0, + "grad_norm": 0.0, + "learning_rate": 0.00981476846057572, + "loss": 0.0, + "step": 80 + }, + { + "epoch": 11.25, + "grad_norm": 0.0, + "learning_rate": 0.00978973717146433, + "loss": 0.0, + "step": 90 + }, + { + "epoch": 12.5, + "grad_norm": 0.0, + "learning_rate": 0.009764705882352941, + "loss": 0.0, + "step": 100 + }, + { + "epoch": 13.75, + "grad_norm": 0.0, + "learning_rate": 0.009739674593241552, + "loss": 0.0, + "step": 110 + }, + { + "epoch": 15.0, + "grad_norm": 0.0, + "learning_rate": 0.009714643304130162, + "loss": 0.0, + "step": 120 + }, + { + "epoch": 16.25, + "grad_norm": 0.0, + "learning_rate": 0.009689612015018775, + "loss": 0.0, + "step": 130 + }, + { + "epoch": 17.5, + "grad_norm": 0.0, + "learning_rate": 0.009664580725907385, + "loss": 0.0, + "step": 140 + }, + { + "epoch": 18.75, + "grad_norm": 0.0, + "learning_rate": 0.009639549436795996, + "loss": 0.0, + "step": 150 + }, + { + "epoch": 20.0, + "grad_norm": 0.0, + "learning_rate": 0.009614518147684605, + "loss": 0.0, + "step": 160 + }, + { + "epoch": 21.25, + "grad_norm": 0.0, + "learning_rate": 0.009589486858573217, + "loss": 0.0, + "step": 170 + }, + { + "epoch": 22.5, + "grad_norm": 0.0, + "learning_rate": 0.009564455569461828, + "loss": 0.0, + "step": 180 + }, + { + "epoch": 23.75, + "grad_norm": 0.0, + "learning_rate": 0.009539424280350439, + "loss": 0.0, + "step": 190 + }, + { + "epoch": 25.0, + "grad_norm": 0.0, + "learning_rate": 0.00951439299123905, + "loss": 0.0, + "step": 200 + }, + { + "epoch": 26.25, + "grad_norm": 0.0, + "learning_rate": 0.00948936170212766, + "loss": 0.0, + "step": 210 + }, + { + "epoch": 27.5, + "grad_norm": 0.0, + "learning_rate": 0.00946433041301627, + "loss": 0.0, + "step": 220 + }, + { + "epoch": 28.75, + "grad_norm": 0.0, + "learning_rate": 0.009439299123904881, + "loss": 0.0, + "step": 230 + }, + { + "epoch": 30.0, + "grad_norm": 0.0, + "learning_rate": 0.009414267834793492, + "loss": 0.0, + "step": 240 + }, + { + "epoch": 31.25, + "grad_norm": 0.0, + "learning_rate": 0.009389236545682102, + "loss": 0.0, + "step": 250 + }, + { + "epoch": 32.5, + "grad_norm": 0.0, + "learning_rate": 0.009364205256570713, + "loss": 0.0, + "step": 260 + }, + { + "epoch": 33.75, + "grad_norm": 0.0, + "learning_rate": 0.009339173967459325, + "loss": 0.0, + "step": 270 + }, + { + "epoch": 35.0, + "grad_norm": 0.0, + "learning_rate": 0.009314142678347936, + "loss": 0.0, + "step": 280 + }, + { + "epoch": 36.25, + "grad_norm": 0.0, + "learning_rate": 0.009289111389236547, + "loss": 0.0, + "step": 290 + }, + { + "epoch": 37.5, + "grad_norm": 0.0, + "learning_rate": 0.009264080100125156, + "loss": 0.0, + "step": 300 + }, + { + "epoch": 38.75, + "grad_norm": 0.0, + "learning_rate": 0.009239048811013768, + "loss": 0.0, + "step": 310 + }, + { + "epoch": 40.0, + "grad_norm": 0.0, + "learning_rate": 0.009214017521902379, + "loss": 0.0, + "step": 320 + }, + { + "epoch": 41.25, + "grad_norm": 0.0, + "learning_rate": 0.00918898623279099, + "loss": 0.0, + "step": 330 + }, + { + "epoch": 42.5, + "grad_norm": 0.0, + "learning_rate": 0.0091639549436796, + "loss": 0.0, + "step": 340 + }, + { + "epoch": 43.75, + "grad_norm": 0.0, + "learning_rate": 0.00913892365456821, + "loss": 0.0, + "step": 350 + }, + { + "epoch": 45.0, + "grad_norm": 0.0, + "learning_rate": 0.009113892365456821, + "loss": 0.0, + "step": 360 + }, + { + "epoch": 46.25, + "grad_norm": 0.0, + "learning_rate": 0.009088861076345432, + "loss": 0.0, + "step": 370 + }, + { + "epoch": 47.5, + "grad_norm": 0.0, + "learning_rate": 0.009063829787234043, + "loss": 0.0, + "step": 380 + }, + { + "epoch": 48.75, + "grad_norm": 0.0, + "learning_rate": 0.009038798498122653, + "loss": 0.0, + "step": 390 + }, + { + "epoch": 50.0, + "grad_norm": 0.0, + "learning_rate": 0.009013767209011264, + "loss": 0.0, + "step": 400 + }, + { + "epoch": 51.25, + "grad_norm": 0.0, + "learning_rate": 0.008988735919899874, + "loss": 0.0, + "step": 410 + }, + { + "epoch": 52.5, + "grad_norm": 0.0, + "learning_rate": 0.008963704630788487, + "loss": 0.0, + "step": 420 + }, + { + "epoch": 53.75, + "grad_norm": 0.0, + "learning_rate": 0.008938673341677096, + "loss": 0.0, + "step": 430 + }, + { + "epoch": 55.0, + "grad_norm": 0.0, + "learning_rate": 0.008913642052565706, + "loss": 0.0, + "step": 440 + }, + { + "epoch": 56.25, + "grad_norm": 0.0, + "learning_rate": 0.008888610763454317, + "loss": 0.0, + "step": 450 + }, + { + "epoch": 57.5, + "grad_norm": 0.0, + "learning_rate": 0.00886357947434293, + "loss": 0.0, + "step": 460 + }, + { + "epoch": 58.75, + "grad_norm": 0.0, + "learning_rate": 0.00883854818523154, + "loss": 0.0, + "step": 470 + }, + { + "epoch": 60.0, + "grad_norm": 0.0, + "learning_rate": 0.00881351689612015, + "loss": 0.0, + "step": 480 + }, + { + "epoch": 61.25, + "grad_norm": 0.0, + "learning_rate": 0.008788485607008761, + "loss": 0.0, + "step": 490 + }, + { + "epoch": 62.5, + "grad_norm": 0.0, + "learning_rate": 0.008763454317897372, + "loss": 0.0, + "step": 500 + }, + { + "epoch": 63.75, + "grad_norm": 0.0, + "learning_rate": 0.008738423028785983, + "loss": 0.0, + "step": 510 + }, + { + "epoch": 65.0, + "grad_norm": 0.0, + "learning_rate": 0.008713391739674593, + "loss": 0.0, + "step": 520 + }, + { + "epoch": 66.25, + "grad_norm": 0.0, + "learning_rate": 0.008688360450563204, + "loss": 0.0, + "step": 530 + }, + { + "epoch": 67.5, + "grad_norm": 0.0, + "learning_rate": 0.008663329161451815, + "loss": 0.0, + "step": 540 + }, + { + "epoch": 68.75, + "grad_norm": 0.0, + "learning_rate": 0.008638297872340425, + "loss": 0.0, + "step": 550 + }, + { + "epoch": 70.0, + "grad_norm": 0.0, + "learning_rate": 0.008613266583229038, + "loss": 0.0, + "step": 560 + }, + { + "epoch": 71.25, + "grad_norm": 0.0, + "learning_rate": 0.008588235294117647, + "loss": 0.0, + "step": 570 + }, + { + "epoch": 72.5, + "grad_norm": 0.0, + "learning_rate": 0.008563204005006257, + "loss": 0.0, + "step": 580 + }, + { + "epoch": 73.75, + "grad_norm": 0.0, + "learning_rate": 0.008538172715894868, + "loss": 0.0, + "step": 590 + }, + { + "epoch": 75.0, + "grad_norm": 0.0, + "learning_rate": 0.00851314142678348, + "loss": 0.0, + "step": 600 + }, + { + "epoch": 76.25, + "grad_norm": 0.0, + "learning_rate": 0.00848811013767209, + "loss": 0.0, + "step": 610 + }, + { + "epoch": 77.5, + "grad_norm": 0.0, + "learning_rate": 0.008463078848560701, + "loss": 0.0, + "step": 620 + }, + { + "epoch": 78.75, + "grad_norm": 0.0, + "learning_rate": 0.008438047559449312, + "loss": 0.0, + "step": 630 + }, + { + "epoch": 80.0, + "grad_norm": 0.0, + "learning_rate": 0.008413016270337923, + "loss": 0.0, + "step": 640 + }, + { + "epoch": 81.25, + "grad_norm": 0.0, + "learning_rate": 0.008387984981226533, + "loss": 0.0, + "step": 650 + }, + { + "epoch": 82.5, + "grad_norm": 0.0, + "learning_rate": 0.008362953692115144, + "loss": 0.0, + "step": 660 + }, + { + "epoch": 83.75, + "grad_norm": 0.0, + "learning_rate": 0.008337922403003755, + "loss": 0.0, + "step": 670 + }, + { + "epoch": 85.0, + "grad_norm": 0.0, + "learning_rate": 0.008312891113892365, + "loss": 0.0, + "step": 680 + }, + { + "epoch": 86.25, + "grad_norm": 0.0, + "learning_rate": 0.008287859824780976, + "loss": 0.0, + "step": 690 + }, + { + "epoch": 87.5, + "grad_norm": 0.0, + "learning_rate": 0.008262828535669588, + "loss": 0.0, + "step": 700 + }, + { + "epoch": 88.75, + "grad_norm": 0.0, + "learning_rate": 0.008237797246558197, + "loss": 0.0, + "step": 710 + }, + { + "epoch": 90.0, + "grad_norm": 0.0, + "learning_rate": 0.008212765957446808, + "loss": 0.0, + "step": 720 + }, + { + "epoch": 91.25, + "grad_norm": 0.0, + "learning_rate": 0.008187734668335419, + "loss": 0.0, + "step": 730 + }, + { + "epoch": 92.5, + "grad_norm": 0.0, + "learning_rate": 0.008162703379224031, + "loss": 0.0, + "step": 740 + }, + { + "epoch": 93.75, + "grad_norm": 0.0, + "learning_rate": 0.008137672090112642, + "loss": 0.0, + "step": 750 + }, + { + "epoch": 95.0, + "grad_norm": 0.0, + "learning_rate": 0.008112640801001252, + "loss": 0.0, + "step": 760 + }, + { + "epoch": 96.25, + "grad_norm": 0.0, + "learning_rate": 0.008087609511889863, + "loss": 0.0, + "step": 770 + }, + { + "epoch": 97.5, + "grad_norm": 0.0, + "learning_rate": 0.008062578222778474, + "loss": 0.0, + "step": 780 + }, + { + "epoch": 98.75, + "grad_norm": 0.0, + "learning_rate": 0.008037546933667084, + "loss": 0.0, + "step": 790 + }, + { + "epoch": 100.0, + "grad_norm": 0.0, + "learning_rate": 0.008012515644555695, + "loss": 0.0, + "step": 800 + }, + { + "epoch": 101.25, + "grad_norm": 0.0, + "learning_rate": 0.007987484355444305, + "loss": 0.0, + "step": 810 + }, + { + "epoch": 102.5, + "grad_norm": 0.0, + "learning_rate": 0.007962453066332916, + "loss": 0.0, + "step": 820 + }, + { + "epoch": 103.75, + "grad_norm": 0.0, + "learning_rate": 0.007937421777221527, + "loss": 0.0, + "step": 830 + }, + { + "epoch": 105.0, + "grad_norm": 0.0, + "learning_rate": 0.007912390488110137, + "loss": 0.0, + "step": 840 + }, + { + "epoch": 106.25, + "grad_norm": 0.0, + "learning_rate": 0.007887359198998748, + "loss": 0.0, + "step": 850 + }, + { + "epoch": 107.5, + "grad_norm": 0.0, + "learning_rate": 0.007862327909887359, + "loss": 0.0, + "step": 860 + }, + { + "epoch": 108.75, + "grad_norm": 0.0, + "learning_rate": 0.00783729662077597, + "loss": 0.0, + "step": 870 + }, + { + "epoch": 110.0, + "grad_norm": 0.0, + "learning_rate": 0.007812265331664581, + "loss": 0.0, + "step": 880 + }, + { + "epoch": 111.25, + "grad_norm": 0.0, + "learning_rate": 0.0077872340425531915, + "loss": 0.0, + "step": 890 + }, + { + "epoch": 112.5, + "grad_norm": 0.0, + "learning_rate": 0.007762202753441803, + "loss": 0.0, + "step": 900 + }, + { + "epoch": 113.75, + "grad_norm": 0.0, + "learning_rate": 0.007737171464330414, + "loss": 0.0, + "step": 910 + }, + { + "epoch": 115.0, + "grad_norm": 0.0, + "learning_rate": 0.0077121401752190235, + "loss": 0.0, + "step": 920 + }, + { + "epoch": 116.25, + "grad_norm": 0.0, + "learning_rate": 0.007687108886107634, + "loss": 0.0, + "step": 930 + }, + { + "epoch": 117.5, + "grad_norm": 0.0, + "learning_rate": 0.007662077596996246, + "loss": 0.0, + "step": 940 + }, + { + "epoch": 118.75, + "grad_norm": 0.0, + "learning_rate": 0.007637046307884856, + "loss": 0.0, + "step": 950 + } + ], + "logging_steps": 10, + "max_steps": 4000, + "num_input_tokens_seen": 0, + "num_train_epochs": 500, + "save_steps": 50, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6413472.0, + "train_batch_size": 64, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-950/training_args.bin b/checkpoint-950/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/checkpoint-950/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137 diff --git a/config.json b/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d827169b4f892e166b1369e7f5c82db745c66c50 --- /dev/null +++ b/config.json @@ -0,0 +1,34 @@ +{ + "activation_function": "gelu", + "add_cross_attention": false, + "architectures": [ + "GPT2LMHeadModel" + ], + "attn_pdrop": 0.0, + "bos_token_id": 1, + "dtype": "float32", + "embd_pdrop": 0.0, + "eos_token_id": 1, + "initializer_range": 0.02, + "layer_norm_epsilon": 1e-05, + "model_type": "gpt2", + "n_embd": 1, + "n_head": 1, + "n_inner": 1, + "n_layer": 1, + "n_positions": 1, + "pad_token_id": 0, + "reorder_and_upcast_attn": false, + "resid_pdrop": 0.0, + "scale_attn_by_inverse_layer_idx": false, + "scale_attn_weights": true, + "summary_activation": null, + "summary_first_dropout": 0.1, + "summary_proj_to_labels": true, + "summary_type": "cls_index", + "summary_use_proj": true, + "tie_word_embeddings": true, + "transformers_version": "5.12.0", + "use_cache": false, + "vocab_size": 2 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ed9bbf4e50a1caae3acc79804c598d0d85cab4df --- /dev/null +++ b/generation_config.json @@ -0,0 +1,10 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 1, + "output_attentions": false, + "output_hidden_states": false, + "pad_token_id": 0, + "transformers_version": "5.12.0", + "use_cache": true +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0be00668724053b639969afc23abf16c826ceb80 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:880c2ec6c05c1e38595824792a1d3850592746c746ee2caa95bee8cb0e8ec77d +size 1452 diff --git a/training_args.bin b/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..66018b5fd33025fea92ff5c3be024140fbcd425b --- /dev/null +++ b/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6208b173b902d75ab9814341e70009f66ecc0ecd70d024038fcaaf650010173 +size 5137