| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.0945945945945945, |
| "eval_steps": 27, |
| "global_step": 243, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.02252252252252252, |
| "grad_norm": 7.59375, |
| "learning_rate": 5.563236689860267e-06, |
| "loss": 1.2494, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.04504504504504504, |
| "grad_norm": 4.3125, |
| "learning_rate": 1.2517282552185602e-05, |
| "loss": 1.0624, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.06756756756756757, |
| "grad_norm": 3.5, |
| "learning_rate": 1.9471328414510936e-05, |
| "loss": 1.0056, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.09009009009009009, |
| "grad_norm": 3.046875, |
| "learning_rate": 1.9467215931370622e-05, |
| "loss": 0.9945, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.11261261261261261, |
| "grad_norm": 3.15625, |
| "learning_rate": 1.9454883114406795e-05, |
| "loss": 0.968, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.13513513513513514, |
| "grad_norm": 2.890625, |
| "learning_rate": 1.9434343855772598e-05, |
| "loss": 0.9378, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.15765765765765766, |
| "grad_norm": 2.984375, |
| "learning_rate": 1.94056212916686e-05, |
| "loss": 0.9841, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.18018018018018017, |
| "grad_norm": 2.9375, |
| "learning_rate": 1.936874777628124e-05, |
| "loss": 0.9823, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.20270270270270271, |
| "grad_norm": 3.09375, |
| "learning_rate": 1.9323764845337886e-05, |
| "loss": 0.9658, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.22522522522522523, |
| "grad_norm": 3.0625, |
| "learning_rate": 1.9270723169319414e-05, |
| "loss": 0.9537, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.24774774774774774, |
| "grad_norm": 2.84375, |
| "learning_rate": 1.9209682496383036e-05, |
| "loss": 0.9701, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.2702702702702703, |
| "grad_norm": 2.859375, |
| "learning_rate": 1.9140711585059715e-05, |
| "loss": 0.9487, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.2927927927927928, |
| "grad_norm": 2.890625, |
| "learning_rate": 1.906388812680195e-05, |
| "loss": 0.9413, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.3153153153153153, |
| "grad_norm": 3.0625, |
| "learning_rate": 1.8979298658469174e-05, |
| "loss": 0.9397, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.33783783783783783, |
| "grad_norm": 2.953125, |
| "learning_rate": 1.8887038464849345e-05, |
| "loss": 0.9395, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.36036036036036034, |
| "grad_norm": 2.796875, |
| "learning_rate": 1.8787211471326552e-05, |
| "loss": 0.9579, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.38288288288288286, |
| "grad_norm": 2.953125, |
| "learning_rate": 1.8679930126815516e-05, |
| "loss": 0.9555, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.40540540540540543, |
| "grad_norm": 2.90625, |
| "learning_rate": 1.8565315277094856e-05, |
| "loss": 0.9535, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.42792792792792794, |
| "grad_norm": 2.875, |
| "learning_rate": 1.8443496028681823e-05, |
| "loss": 0.9553, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.45045045045045046, |
| "grad_norm": 2.859375, |
| "learning_rate": 1.8314609603401823e-05, |
| "loss": 0.943, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.47297297297297297, |
| "grad_norm": 3.5625, |
| "learning_rate": 1.817880118381654e-05, |
| "loss": 0.9296, |
| "step": 105 |
| }, |
| { |
| "epoch": 0.4954954954954955, |
| "grad_norm": 2.71875, |
| "learning_rate": 1.8036223749684777e-05, |
| "loss": 0.9294, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.5180180180180181, |
| "grad_norm": 2.65625, |
| "learning_rate": 1.7887037905640276e-05, |
| "loss": 0.9259, |
| "step": 115 |
| }, |
| { |
| "epoch": 0.5405405405405406, |
| "grad_norm": 2.6875, |
| "learning_rate": 1.773141170028053e-05, |
| "loss": 0.9304, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.5630630630630631, |
| "grad_norm": 2.75, |
| "learning_rate": 1.7569520436870487e-05, |
| "loss": 0.9245, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.5855855855855856, |
| "grad_norm": 2.78125, |
| "learning_rate": 1.7401546475874292e-05, |
| "loss": 0.9355, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.6081081081081081, |
| "grad_norm": 2.890625, |
| "learning_rate": 1.7227679029537527e-05, |
| "loss": 0.9317, |
| "step": 135 |
| }, |
| { |
| "epoch": 0.6306306306306306, |
| "grad_norm": 2.953125, |
| "learning_rate": 1.7048113948751378e-05, |
| "loss": 0.9088, |
| "step": 140 |
| }, |
| { |
| "epoch": 0.6531531531531531, |
| "grad_norm": 2.703125, |
| "learning_rate": 1.6863053502438737e-05, |
| "loss": 0.9042, |
| "step": 145 |
| }, |
| { |
| "epoch": 0.6756756756756757, |
| "grad_norm": 2.765625, |
| "learning_rate": 1.6672706149710824e-05, |
| "loss": 0.919, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.6981981981981982, |
| "grad_norm": 2.65625, |
| "learning_rate": 1.647728630505091e-05, |
| "loss": 0.9464, |
| "step": 155 |
| }, |
| { |
| "epoch": 0.7207207207207207, |
| "grad_norm": 2.71875, |
| "learning_rate": 1.6277014096789735e-05, |
| "loss": 0.9213, |
| "step": 160 |
| }, |
| { |
| "epoch": 0.7432432432432432, |
| "grad_norm": 2.703125, |
| "learning_rate": 1.607211511914458e-05, |
| "loss": 0.9444, |
| "step": 165 |
| }, |
| { |
| "epoch": 0.7657657657657657, |
| "grad_norm": 2.8125, |
| "learning_rate": 1.586282017810137e-05, |
| "loss": 0.9435, |
| "step": 170 |
| }, |
| { |
| "epoch": 0.7882882882882883, |
| "grad_norm": 2.890625, |
| "learning_rate": 1.5649365031426085e-05, |
| "loss": 0.9189, |
| "step": 175 |
| }, |
| { |
| "epoch": 0.8108108108108109, |
| "grad_norm": 2.875, |
| "learning_rate": 1.543199012309822e-05, |
| "loss": 0.9483, |
| "step": 180 |
| }, |
| { |
| "epoch": 0.8333333333333334, |
| "grad_norm": 2.953125, |
| "learning_rate": 1.5210940312465595e-05, |
| "loss": 0.9302, |
| "step": 185 |
| }, |
| { |
| "epoch": 0.8558558558558559, |
| "grad_norm": 2.71875, |
| "learning_rate": 1.4986464598425472e-05, |
| "loss": 0.9173, |
| "step": 190 |
| }, |
| { |
| "epoch": 0.8783783783783784, |
| "grad_norm": 2.90625, |
| "learning_rate": 1.4758815838942745e-05, |
| "loss": 0.9389, |
| "step": 195 |
| }, |
| { |
| "epoch": 0.9009009009009009, |
| "grad_norm": 2.609375, |
| "learning_rate": 1.4528250466221154e-05, |
| "loss": 0.9006, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.9234234234234234, |
| "grad_norm": 2.84375, |
| "learning_rate": 1.4295028197848276e-05, |
| "loss": 0.9245, |
| "step": 205 |
| }, |
| { |
| "epoch": 0.9459459459459459, |
| "grad_norm": 3.53125, |
| "learning_rate": 1.4059411744239822e-05, |
| "loss": 0.904, |
| "step": 210 |
| }, |
| { |
| "epoch": 0.9684684684684685, |
| "grad_norm": 2.671875, |
| "learning_rate": 1.3821666512712616e-05, |
| "loss": 0.9093, |
| "step": 215 |
| }, |
| { |
| "epoch": 0.990990990990991, |
| "grad_norm": 2.71875, |
| "learning_rate": 1.3582060308519734e-05, |
| "loss": 0.9232, |
| "step": 220 |
| }, |
| { |
| "epoch": 1.0135135135135136, |
| "grad_norm": 3.15625, |
| "learning_rate": 1.3340863033184511e-05, |
| "loss": 0.7718, |
| "step": 225 |
| }, |
| { |
| "epoch": 1.0360360360360361, |
| "grad_norm": 3.875, |
| "learning_rate": 1.3098346380473199e-05, |
| "loss": 0.6475, |
| "step": 230 |
| }, |
| { |
| "epoch": 1.0585585585585586, |
| "grad_norm": 3.265625, |
| "learning_rate": 1.2854783530348815e-05, |
| "loss": 0.6412, |
| "step": 235 |
| }, |
| { |
| "epoch": 1.0810810810810811, |
| "grad_norm": 3.109375, |
| "learning_rate": 1.2610448841250846e-05, |
| "loss": 0.6408, |
| "step": 240 |
| }, |
| { |
| "epoch": 1.0945945945945945, |
| "eval_loss": 0.9434097409248352, |
| "eval_runtime": 9.493, |
| "eval_samples_per_second": 26.23, |
| "eval_steps_per_second": 13.168, |
| "step": 243 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 482, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 3, |
| "save_steps": 27, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 2.2689703520527974e+17, |
| "train_batch_size": 18, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|