{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 2.736842105263158, "eval_steps": 500, "global_step": 27, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.10526315789473684, "grad_norm": 66.80625915527344, "learning_rate": 0.0003, "loss": 59.0922, "step": 1 }, { "epoch": 0.21052631578947367, "grad_norm": 42.16349411010742, "learning_rate": 0.0003, "loss": 40.5625, "step": 2 }, { "epoch": 0.3157894736842105, "grad_norm": 34.7736930847168, "learning_rate": 0.0003, "loss": 37.6795, "step": 3 }, { "epoch": 0.42105263157894735, "grad_norm": 41.49787139892578, "learning_rate": 0.0003, "loss": 23.9383, "step": 4 }, { "epoch": 0.5263157894736842, "grad_norm": 52.929927825927734, "learning_rate": 0.0003, "loss": 19.2318, "step": 5 }, { "epoch": 0.631578947368421, "grad_norm": 35.0192756652832, "learning_rate": 0.0003, "loss": 24.5318, "step": 6 }, { "epoch": 0.7368421052631579, "grad_norm": 22.34128761291504, "learning_rate": 0.0003, "loss": 20.694, "step": 7 }, { "epoch": 0.8421052631578947, "grad_norm": 33.19445037841797, "learning_rate": 0.0003, "loss": 15.9587, "step": 8 }, { "epoch": 0.9473684210526315, "grad_norm": 35.2607307434082, "learning_rate": 0.0003, "loss": 16.091, "step": 9 }, { "epoch": 1.0, "grad_norm": 17.971670150756836, "learning_rate": 0.0003, "loss": 6.4848, "step": 10 }, { "epoch": 1.1052631578947367, "grad_norm": 70.17874908447266, "learning_rate": 0.0003, "loss": 21.502, "step": 11 }, { "epoch": 1.2105263157894737, "grad_norm": 9.148673057556152, "learning_rate": 0.0003, "loss": 9.0273, "step": 12 }, { "epoch": 1.3157894736842106, "grad_norm": 8.400835990905762, "learning_rate": 0.0003, "loss": 14.5862, "step": 13 }, { "epoch": 1.4210526315789473, "grad_norm": 8.655104637145996, "learning_rate": 0.0003, "loss": 14.0817, "step": 14 }, { "epoch": 1.526315789473684, "grad_norm": 8.041231155395508, "learning_rate": 0.0003, "loss": 19.4253, "step": 15 }, { "epoch": 1.631578947368421, "grad_norm": 11.80112361907959, "learning_rate": 0.0003, "loss": 10.6932, "step": 16 }, { "epoch": 1.736842105263158, "grad_norm": 24.622623443603516, "learning_rate": 0.0003, "loss": 8.2083, "step": 17 }, { "epoch": 1.8421052631578947, "grad_norm": 12.404068946838379, "learning_rate": 0.0003, "loss": 16.6983, "step": 18 }, { "epoch": 1.9473684210526314, "grad_norm": 11.858294486999512, "learning_rate": 0.0003, "loss": 18.4896, "step": 19 }, { "epoch": 2.0, "grad_norm": 10.483745574951172, "learning_rate": 0.0003, "loss": 4.207, "step": 20 }, { "epoch": 2.1052631578947367, "grad_norm": 8.02437973022461, "learning_rate": 0.0003, "loss": 14.6837, "step": 21 }, { "epoch": 2.2105263157894735, "grad_norm": 11.133962631225586, "learning_rate": 0.0003, "loss": 11.7229, "step": 22 }, { "epoch": 2.3157894736842106, "grad_norm": 14.599047660827637, "learning_rate": 0.0003, "loss": 11.4597, "step": 23 }, { "epoch": 2.4210526315789473, "grad_norm": 12.829426765441895, "learning_rate": 0.0003, "loss": 9.1055, "step": 24 }, { "epoch": 2.526315789473684, "grad_norm": 6.030374050140381, "learning_rate": 0.0003, "loss": 11.2372, "step": 25 }, { "epoch": 2.6315789473684212, "grad_norm": 12.609663009643555, "learning_rate": 0.0003, "loss": 8.3892, "step": 26 }, { "epoch": 2.736842105263158, "grad_norm": 6.034035682678223, "learning_rate": 0.0003, "loss": 10.5367, "step": 27 }, { "epoch": 2.736842105263158, "step": 27, "total_flos": 0.0, "train_loss": 17.715496469427038, "train_runtime": 155.5004, "train_samples_per_second": 5.788, "train_steps_per_second": 0.174 } ], "logging_steps": 1, "max_steps": 27, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }