{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 1000, "global_step": 2000, "is_hyper_param_search": false, "is_local_process_zero": false, "is_world_process_zero": true, "log_history": [ { "epoch": 0.05, "grad_norm": 1.485682487487793, "learning_rate": 0.003980291252282261, "loss": 73.7283203125, "step": 100 }, { "epoch": 0.1, "grad_norm": 1.4347572326660156, "learning_rate": 0.003911632443486905, "loss": 67.1016357421875, "step": 200 }, { "epoch": 0.15, "grad_norm": 1.7708789110183716, "learning_rate": 0.003795429623632202, "loss": 65.27525390625, "step": 300 }, { "epoch": 0.2, "grad_norm": 1.349237322807312, "learning_rate": 0.0036345728609271967, "loss": 64.15083984375, "step": 400 }, { "epoch": 0.25, "grad_norm": 1.3039754629135132, "learning_rate": 0.003433062807134769, "loss": 63.8335498046875, "step": 500 }, { "epoch": 0.3, "grad_norm": 1.2484228610992432, "learning_rate": 0.0031959111977790367, "loss": 63.246728515625, "step": 600 }, { "epoch": 0.35, "grad_norm": 1.4882639646530151, "learning_rate": 0.0029290162057940094, "loss": 62.8143017578125, "step": 700 }, { "epoch": 0.4, "grad_norm": 1.1007429361343384, "learning_rate": 0.0026390157486799134, "loss": 62.51708984375, "step": 800 }, { "epoch": 0.45, "grad_norm": 0.9959006905555725, "learning_rate": 0.0023331223975495813, "loss": 62.166494140625, "step": 900 }, { "epoch": 0.5, "grad_norm": 1.243346929550171, "learning_rate": 0.002018943994024803, "loss": 62.0639892578125, "step": 1000 }, { "epoch": 0.5, "eval_accuracy": 0.15881036168132942, "eval_loss": 61.933006286621094, "eval_runtime": 21.6587, "eval_samples_per_second": 5.771, "eval_steps_per_second": 0.185, "step": 1000 }, { "epoch": 0.55, "grad_norm": 1.3841041326522827, "learning_rate": 0.0017042944364010796, "loss": 61.8725634765625, "step": 1100 }, { "epoch": 0.6, "grad_norm": 0.9040701389312744, "learning_rate": 0.0013969993409983545, "loss": 61.6279736328125, "step": 1200 }, { "epoch": 0.65, "grad_norm": 0.9380832314491272, "learning_rate": 0.0011047014120739685, "loss": 61.5527197265625, "step": 1300 }, { "epoch": 0.7, "grad_norm": 0.8851712346076965, "learning_rate": 0.0008346703609224516, "loss": 61.3382958984375, "step": 1400 }, { "epoch": 0.75, "grad_norm": 1.1163753271102905, "learning_rate": 0.0005936221016443706, "loss": 61.1475, "step": 1500 }, { "epoch": 0.8, "grad_norm": 0.8482072353363037, "learning_rate": 0.0003875517203474137, "loss": 61.2019091796875, "step": 1600 }, { "epoch": 0.85, "grad_norm": 0.7953028082847595, "learning_rate": 0.00022158437198527747, "loss": 61.2163671875, "step": 1700 }, { "epoch": 0.9, "grad_norm": 0.8152965903282166, "learning_rate": 9.984781316353054e-05, "loss": 60.9908349609375, "step": 1800 }, { "epoch": 0.95, "grad_norm": 0.8798665404319763, "learning_rate": 2.536974113572521e-05, "loss": 61.030517578125, "step": 1900 }, { "epoch": 1.0, "grad_norm": 0.9654077291488647, "learning_rate": 2.4922608901079e-09, "loss": 61.0981298828125, "step": 2000 }, { "epoch": 1.0, "eval_accuracy": 0.1659012707722385, "eval_loss": 61.03606033325195, "eval_runtime": 6.2006, "eval_samples_per_second": 20.159, "eval_steps_per_second": 0.645, "step": 2000 } ], "logging_steps": 100, "max_steps": 2000, "num_input_tokens_seen": 0, "num_train_epochs": 9223372036854775807, "save_steps": 2000, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }