| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.06, |
| "eval_steps": 150, |
| "global_step": 300, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.001, |
| "grad_norm": 0.27235767245292664, |
| "learning_rate": 6.4e-08, |
| "loss": 10.786966705322266, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.002, |
| "grad_norm": 0.36470845341682434, |
| "learning_rate": 1.44e-07, |
| "loss": 10.004191589355468, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.003, |
| "grad_norm": 0.18794603645801544, |
| "learning_rate": 2.24e-07, |
| "loss": 9.357711791992188, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 0.15133747458457947, |
| "learning_rate": 3.0399999999999997e-07, |
| "loss": 9.077755737304688, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.005, |
| "grad_norm": 0.13880157470703125, |
| "learning_rate": 3.84e-07, |
| "loss": 8.826399993896484, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 0.12539362907409668, |
| "learning_rate": 4.64e-07, |
| "loss": 8.724310302734375, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.007, |
| "grad_norm": 0.11178825795650482, |
| "learning_rate": 5.44e-07, |
| "loss": 8.595573425292969, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 0.16203832626342773, |
| "learning_rate": 6.24e-07, |
| "loss": 8.613536834716797, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.009, |
| "grad_norm": 0.11954408138990402, |
| "learning_rate": 7.04e-07, |
| "loss": 8.435651397705078, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 0.09399693459272385, |
| "learning_rate": 7.84e-07, |
| "loss": 8.369234466552735, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.011, |
| "grad_norm": 0.08723891526460648, |
| "learning_rate": 8.639999999999999e-07, |
| "loss": 8.335379791259765, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 0.09454073011875153, |
| "learning_rate": 9.439999999999999e-07, |
| "loss": 8.253787231445312, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.013, |
| "grad_norm": 0.09540031850337982, |
| "learning_rate": 1.024e-06, |
| "loss": 8.186421966552734, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 0.11653102189302444, |
| "learning_rate": 1.1040000000000001e-06, |
| "loss": 8.132817840576172, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.015, |
| "grad_norm": 0.08094990253448486, |
| "learning_rate": 1.1839999999999998e-06, |
| "loss": 8.107611846923827, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 0.10092103481292725, |
| "learning_rate": 1.2639999999999999e-06, |
| "loss": 8.071889495849609, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.017, |
| "grad_norm": 0.09236462414264679, |
| "learning_rate": 1.344e-06, |
| "loss": 8.15390625, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 0.08973367512226105, |
| "learning_rate": 1.4239999999999998e-06, |
| "loss": 8.073948669433594, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.019, |
| "grad_norm": 0.08046946674585342, |
| "learning_rate": 1.504e-06, |
| "loss": 7.942198944091797, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 0.07841245085000992, |
| "learning_rate": 1.584e-06, |
| "loss": 7.894329071044922, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.021, |
| "grad_norm": 0.07948298752307892, |
| "learning_rate": 1.6639999999999999e-06, |
| "loss": 7.908601379394531, |
| "step": 105 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 0.07720845937728882, |
| "learning_rate": 1.744e-06, |
| "loss": 7.923919677734375, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.023, |
| "grad_norm": 0.09289873391389847, |
| "learning_rate": 1.824e-06, |
| "loss": 7.866429138183594, |
| "step": 115 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 0.07957543432712555, |
| "learning_rate": 1.904e-06, |
| "loss": 7.828727722167969, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.025, |
| "grad_norm": 0.07522212713956833, |
| "learning_rate": 1.984e-06, |
| "loss": 7.83117446899414, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 0.09532520920038223, |
| "learning_rate": 2.064e-06, |
| "loss": 7.8716987609863285, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.027, |
| "grad_norm": 0.07470740377902985, |
| "learning_rate": 2.144e-06, |
| "loss": 7.834799194335938, |
| "step": 135 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 0.07192815095186234, |
| "learning_rate": 2.2240000000000002e-06, |
| "loss": 7.736669921875, |
| "step": 140 |
| }, |
| { |
| "epoch": 0.029, |
| "grad_norm": 0.09201541543006897, |
| "learning_rate": 2.304e-06, |
| "loss": 7.6827842712402346, |
| "step": 145 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 0.09225732833147049, |
| "learning_rate": 2.384e-06, |
| "loss": 7.745516967773438, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.03, |
| "eval_accuracy": 0.11922344322344322, |
| "eval_loss": 7.651349067687988, |
| "eval_runtime": 10.4261, |
| "eval_samples_per_second": 9.591, |
| "eval_steps_per_second": 1.631, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.031, |
| "grad_norm": 0.08881130814552307, |
| "learning_rate": 2.464e-06, |
| "loss": 7.674343872070312, |
| "step": 155 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 0.07704972475767136, |
| "learning_rate": 2.544e-06, |
| "loss": 7.681053161621094, |
| "step": 160 |
| }, |
| { |
| "epoch": 0.033, |
| "grad_norm": 0.09186960011720657, |
| "learning_rate": 2.624e-06, |
| "loss": 7.690374755859375, |
| "step": 165 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 0.08565036207437515, |
| "learning_rate": 2.704e-06, |
| "loss": 7.603900146484375, |
| "step": 170 |
| }, |
| { |
| "epoch": 0.035, |
| "grad_norm": 0.1499108374118805, |
| "learning_rate": 2.7839999999999995e-06, |
| "loss": 7.505958557128906, |
| "step": 175 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 0.06643059849739075, |
| "learning_rate": 2.8639999999999996e-06, |
| "loss": 7.573754119873047, |
| "step": 180 |
| }, |
| { |
| "epoch": 0.037, |
| "grad_norm": 0.11821448802947998, |
| "learning_rate": 2.9439999999999997e-06, |
| "loss": 7.752062225341797, |
| "step": 185 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 0.08827198296785355, |
| "learning_rate": 3.0239999999999998e-06, |
| "loss": 7.5669189453125, |
| "step": 190 |
| }, |
| { |
| "epoch": 0.039, |
| "grad_norm": 0.07519616931676865, |
| "learning_rate": 3.104e-06, |
| "loss": 7.571305847167968, |
| "step": 195 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 0.075102798640728, |
| "learning_rate": 3.184e-06, |
| "loss": 7.518638610839844, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.041, |
| "grad_norm": 0.10685888677835464, |
| "learning_rate": 3.2639999999999996e-06, |
| "loss": 7.512000274658203, |
| "step": 205 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 0.10541684180498123, |
| "learning_rate": 3.3439999999999997e-06, |
| "loss": 7.564778137207031, |
| "step": 210 |
| }, |
| { |
| "epoch": 0.043, |
| "grad_norm": 0.09155019372701645, |
| "learning_rate": 3.4239999999999997e-06, |
| "loss": 7.4674217224121096, |
| "step": 215 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 0.11373615264892578, |
| "learning_rate": 3.504e-06, |
| "loss": 7.450457000732422, |
| "step": 220 |
| }, |
| { |
| "epoch": 0.045, |
| "grad_norm": 0.07444982975721359, |
| "learning_rate": 3.584e-06, |
| "loss": 7.397786712646484, |
| "step": 225 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 0.08052903413772583, |
| "learning_rate": 3.664e-06, |
| "loss": 7.415688323974609, |
| "step": 230 |
| }, |
| { |
| "epoch": 0.047, |
| "grad_norm": 0.08429361879825592, |
| "learning_rate": 3.744e-06, |
| "loss": 7.411205291748047, |
| "step": 235 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 0.08501375466585159, |
| "learning_rate": 3.823999999999999e-06, |
| "loss": 7.406707763671875, |
| "step": 240 |
| }, |
| { |
| "epoch": 0.049, |
| "grad_norm": 0.10457716882228851, |
| "learning_rate": 3.903999999999999e-06, |
| "loss": 7.312602233886719, |
| "step": 245 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 0.09624126553535461, |
| "learning_rate": 3.9839999999999995e-06, |
| "loss": 7.360851287841797, |
| "step": 250 |
| }, |
| { |
| "epoch": 0.051, |
| "grad_norm": 0.08660672605037689, |
| "learning_rate": 3.99999300106024e-06, |
| "loss": 7.353035736083984, |
| "step": 255 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 0.08430126309394836, |
| "learning_rate": 3.9999645679514235e-06, |
| "loss": 7.384159088134766, |
| "step": 260 |
| }, |
| { |
| "epoch": 0.053, |
| "grad_norm": 0.07826503366231918, |
| "learning_rate": 3.999914263550513e-06, |
| "loss": 7.4195198059082035, |
| "step": 265 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 0.08577796816825867, |
| "learning_rate": 3.9998420884076325e-06, |
| "loss": 7.362513732910156, |
| "step": 270 |
| }, |
| { |
| "epoch": 0.055, |
| "grad_norm": 0.0649741068482399, |
| "learning_rate": 3.999748043312075e-06, |
| "loss": 7.275961303710938, |
| "step": 275 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 0.07203389704227448, |
| "learning_rate": 3.999632129292304e-06, |
| "loss": 7.263797760009766, |
| "step": 280 |
| }, |
| { |
| "epoch": 0.057, |
| "grad_norm": 0.12281067669391632, |
| "learning_rate": 3.9994943476159364e-06, |
| "loss": 7.287098693847656, |
| "step": 285 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 0.07184985280036926, |
| "learning_rate": 3.999334699789731e-06, |
| "loss": 7.240232849121094, |
| "step": 290 |
| }, |
| { |
| "epoch": 0.059, |
| "grad_norm": 0.07799801975488663, |
| "learning_rate": 3.99915318755957e-06, |
| "loss": 7.2740531921386715, |
| "step": 295 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 0.0761650875210762, |
| "learning_rate": 3.9989498129104425e-06, |
| "loss": 7.299461364746094, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_accuracy": 0.12382905982905983, |
| "eval_loss": 7.217157363891602, |
| "eval_runtime": 10.4259, |
| "eval_samples_per_second": 9.591, |
| "eval_steps_per_second": 1.631, |
| "step": 300 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 5000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 300, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 6, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|