| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.0, |
| "eval_steps": 1000, |
| "global_step": 2000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": false, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.05, |
| "grad_norm": 1.485682487487793, |
| "learning_rate": 0.003980291252282261, |
| "loss": 73.7283203125, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.1, |
| "grad_norm": 1.4347572326660156, |
| "learning_rate": 0.003911632443486905, |
| "loss": 67.1016357421875, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.15, |
| "grad_norm": 1.7708789110183716, |
| "learning_rate": 0.003795429623632202, |
| "loss": 65.27525390625, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.2, |
| "grad_norm": 1.349237322807312, |
| "learning_rate": 0.0036345728609271967, |
| "loss": 64.15083984375, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.25, |
| "grad_norm": 1.3039754629135132, |
| "learning_rate": 0.003433062807134769, |
| "loss": 63.8335498046875, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.3, |
| "grad_norm": 1.2484228610992432, |
| "learning_rate": 0.0031959111977790367, |
| "loss": 63.246728515625, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.35, |
| "grad_norm": 1.4882639646530151, |
| "learning_rate": 0.0029290162057940094, |
| "loss": 62.8143017578125, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.4, |
| "grad_norm": 1.1007429361343384, |
| "learning_rate": 0.0026390157486799134, |
| "loss": 62.51708984375, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.45, |
| "grad_norm": 0.9959006905555725, |
| "learning_rate": 0.0023331223975495813, |
| "loss": 62.166494140625, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.5, |
| "grad_norm": 1.243346929550171, |
| "learning_rate": 0.002018943994024803, |
| "loss": 62.0639892578125, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.5, |
| "eval_accuracy": 0.15881036168132942, |
| "eval_loss": 61.933006286621094, |
| "eval_runtime": 21.6587, |
| "eval_samples_per_second": 5.771, |
| "eval_steps_per_second": 0.185, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.55, |
| "grad_norm": 1.3841041326522827, |
| "learning_rate": 0.0017042944364010796, |
| "loss": 61.8725634765625, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.6, |
| "grad_norm": 0.9040701389312744, |
| "learning_rate": 0.0013969993409983545, |
| "loss": 61.6279736328125, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.65, |
| "grad_norm": 0.9380832314491272, |
| "learning_rate": 0.0011047014120739685, |
| "loss": 61.5527197265625, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.7, |
| "grad_norm": 0.8851712346076965, |
| "learning_rate": 0.0008346703609224516, |
| "loss": 61.3382958984375, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.75, |
| "grad_norm": 1.1163753271102905, |
| "learning_rate": 0.0005936221016443706, |
| "loss": 61.1475, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.8, |
| "grad_norm": 0.8482072353363037, |
| "learning_rate": 0.0003875517203474137, |
| "loss": 61.2019091796875, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.85, |
| "grad_norm": 0.7953028082847595, |
| "learning_rate": 0.00022158437198527747, |
| "loss": 61.2163671875, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.9, |
| "grad_norm": 0.8152965903282166, |
| "learning_rate": 9.984781316353054e-05, |
| "loss": 60.9908349609375, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.95, |
| "grad_norm": 0.8798665404319763, |
| "learning_rate": 2.536974113572521e-05, |
| "loss": 61.030517578125, |
| "step": 1900 |
| }, |
| { |
| "epoch": 1.0, |
| "grad_norm": 0.9654077291488647, |
| "learning_rate": 2.4922608901079e-09, |
| "loss": 61.0981298828125, |
| "step": 2000 |
| }, |
| { |
| "epoch": 1.0, |
| "eval_accuracy": 0.1659012707722385, |
| "eval_loss": 61.03606033325195, |
| "eval_runtime": 6.2006, |
| "eval_samples_per_second": 20.159, |
| "eval_steps_per_second": 0.645, |
| "step": 2000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 2000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": true |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|