CodeIsAbstract's picture
Training in progress, step 300, checkpoint
22adbad verified
Raw
History Blame
11.2 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.06,
"eval_steps": 150,
"global_step": 300,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.001,
"grad_norm": 0.27235767245292664,
"learning_rate": 6.4e-08,
"loss": 10.786966705322266,
"step": 5
},
{
"epoch": 0.002,
"grad_norm": 0.36470845341682434,
"learning_rate": 1.44e-07,
"loss": 10.004191589355468,
"step": 10
},
{
"epoch": 0.003,
"grad_norm": 0.18794603645801544,
"learning_rate": 2.24e-07,
"loss": 9.357711791992188,
"step": 15
},
{
"epoch": 0.004,
"grad_norm": 0.15133747458457947,
"learning_rate": 3.0399999999999997e-07,
"loss": 9.077755737304688,
"step": 20
},
{
"epoch": 0.005,
"grad_norm": 0.13880157470703125,
"learning_rate": 3.84e-07,
"loss": 8.826399993896484,
"step": 25
},
{
"epoch": 0.006,
"grad_norm": 0.12539362907409668,
"learning_rate": 4.64e-07,
"loss": 8.724310302734375,
"step": 30
},
{
"epoch": 0.007,
"grad_norm": 0.11178825795650482,
"learning_rate": 5.44e-07,
"loss": 8.595573425292969,
"step": 35
},
{
"epoch": 0.008,
"grad_norm": 0.16203832626342773,
"learning_rate": 6.24e-07,
"loss": 8.613536834716797,
"step": 40
},
{
"epoch": 0.009,
"grad_norm": 0.11954408138990402,
"learning_rate": 7.04e-07,
"loss": 8.435651397705078,
"step": 45
},
{
"epoch": 0.01,
"grad_norm": 0.09399693459272385,
"learning_rate": 7.84e-07,
"loss": 8.369234466552735,
"step": 50
},
{
"epoch": 0.011,
"grad_norm": 0.08723891526460648,
"learning_rate": 8.639999999999999e-07,
"loss": 8.335379791259765,
"step": 55
},
{
"epoch": 0.012,
"grad_norm": 0.09454073011875153,
"learning_rate": 9.439999999999999e-07,
"loss": 8.253787231445312,
"step": 60
},
{
"epoch": 0.013,
"grad_norm": 0.09540031850337982,
"learning_rate": 1.024e-06,
"loss": 8.186421966552734,
"step": 65
},
{
"epoch": 0.014,
"grad_norm": 0.11653102189302444,
"learning_rate": 1.1040000000000001e-06,
"loss": 8.132817840576172,
"step": 70
},
{
"epoch": 0.015,
"grad_norm": 0.08094990253448486,
"learning_rate": 1.1839999999999998e-06,
"loss": 8.107611846923827,
"step": 75
},
{
"epoch": 0.016,
"grad_norm": 0.10092103481292725,
"learning_rate": 1.2639999999999999e-06,
"loss": 8.071889495849609,
"step": 80
},
{
"epoch": 0.017,
"grad_norm": 0.09236462414264679,
"learning_rate": 1.344e-06,
"loss": 8.15390625,
"step": 85
},
{
"epoch": 0.018,
"grad_norm": 0.08973367512226105,
"learning_rate": 1.4239999999999998e-06,
"loss": 8.073948669433594,
"step": 90
},
{
"epoch": 0.019,
"grad_norm": 0.08046946674585342,
"learning_rate": 1.504e-06,
"loss": 7.942198944091797,
"step": 95
},
{
"epoch": 0.02,
"grad_norm": 0.07841245085000992,
"learning_rate": 1.584e-06,
"loss": 7.894329071044922,
"step": 100
},
{
"epoch": 0.021,
"grad_norm": 0.07948298752307892,
"learning_rate": 1.6639999999999999e-06,
"loss": 7.908601379394531,
"step": 105
},
{
"epoch": 0.022,
"grad_norm": 0.07720845937728882,
"learning_rate": 1.744e-06,
"loss": 7.923919677734375,
"step": 110
},
{
"epoch": 0.023,
"grad_norm": 0.09289873391389847,
"learning_rate": 1.824e-06,
"loss": 7.866429138183594,
"step": 115
},
{
"epoch": 0.024,
"grad_norm": 0.07957543432712555,
"learning_rate": 1.904e-06,
"loss": 7.828727722167969,
"step": 120
},
{
"epoch": 0.025,
"grad_norm": 0.07522212713956833,
"learning_rate": 1.984e-06,
"loss": 7.83117446899414,
"step": 125
},
{
"epoch": 0.026,
"grad_norm": 0.09532520920038223,
"learning_rate": 2.064e-06,
"loss": 7.8716987609863285,
"step": 130
},
{
"epoch": 0.027,
"grad_norm": 0.07470740377902985,
"learning_rate": 2.144e-06,
"loss": 7.834799194335938,
"step": 135
},
{
"epoch": 0.028,
"grad_norm": 0.07192815095186234,
"learning_rate": 2.2240000000000002e-06,
"loss": 7.736669921875,
"step": 140
},
{
"epoch": 0.029,
"grad_norm": 0.09201541543006897,
"learning_rate": 2.304e-06,
"loss": 7.6827842712402346,
"step": 145
},
{
"epoch": 0.03,
"grad_norm": 0.09225732833147049,
"learning_rate": 2.384e-06,
"loss": 7.745516967773438,
"step": 150
},
{
"epoch": 0.03,
"eval_accuracy": 0.11922344322344322,
"eval_loss": 7.651349067687988,
"eval_runtime": 10.4261,
"eval_samples_per_second": 9.591,
"eval_steps_per_second": 1.631,
"step": 150
},
{
"epoch": 0.031,
"grad_norm": 0.08881130814552307,
"learning_rate": 2.464e-06,
"loss": 7.674343872070312,
"step": 155
},
{
"epoch": 0.032,
"grad_norm": 0.07704972475767136,
"learning_rate": 2.544e-06,
"loss": 7.681053161621094,
"step": 160
},
{
"epoch": 0.033,
"grad_norm": 0.09186960011720657,
"learning_rate": 2.624e-06,
"loss": 7.690374755859375,
"step": 165
},
{
"epoch": 0.034,
"grad_norm": 0.08565036207437515,
"learning_rate": 2.704e-06,
"loss": 7.603900146484375,
"step": 170
},
{
"epoch": 0.035,
"grad_norm": 0.1499108374118805,
"learning_rate": 2.7839999999999995e-06,
"loss": 7.505958557128906,
"step": 175
},
{
"epoch": 0.036,
"grad_norm": 0.06643059849739075,
"learning_rate": 2.8639999999999996e-06,
"loss": 7.573754119873047,
"step": 180
},
{
"epoch": 0.037,
"grad_norm": 0.11821448802947998,
"learning_rate": 2.9439999999999997e-06,
"loss": 7.752062225341797,
"step": 185
},
{
"epoch": 0.038,
"grad_norm": 0.08827198296785355,
"learning_rate": 3.0239999999999998e-06,
"loss": 7.5669189453125,
"step": 190
},
{
"epoch": 0.039,
"grad_norm": 0.07519616931676865,
"learning_rate": 3.104e-06,
"loss": 7.571305847167968,
"step": 195
},
{
"epoch": 0.04,
"grad_norm": 0.075102798640728,
"learning_rate": 3.184e-06,
"loss": 7.518638610839844,
"step": 200
},
{
"epoch": 0.041,
"grad_norm": 0.10685888677835464,
"learning_rate": 3.2639999999999996e-06,
"loss": 7.512000274658203,
"step": 205
},
{
"epoch": 0.042,
"grad_norm": 0.10541684180498123,
"learning_rate": 3.3439999999999997e-06,
"loss": 7.564778137207031,
"step": 210
},
{
"epoch": 0.043,
"grad_norm": 0.09155019372701645,
"learning_rate": 3.4239999999999997e-06,
"loss": 7.4674217224121096,
"step": 215
},
{
"epoch": 0.044,
"grad_norm": 0.11373615264892578,
"learning_rate": 3.504e-06,
"loss": 7.450457000732422,
"step": 220
},
{
"epoch": 0.045,
"grad_norm": 0.07444982975721359,
"learning_rate": 3.584e-06,
"loss": 7.397786712646484,
"step": 225
},
{
"epoch": 0.046,
"grad_norm": 0.08052903413772583,
"learning_rate": 3.664e-06,
"loss": 7.415688323974609,
"step": 230
},
{
"epoch": 0.047,
"grad_norm": 0.08429361879825592,
"learning_rate": 3.744e-06,
"loss": 7.411205291748047,
"step": 235
},
{
"epoch": 0.048,
"grad_norm": 0.08501375466585159,
"learning_rate": 3.823999999999999e-06,
"loss": 7.406707763671875,
"step": 240
},
{
"epoch": 0.049,
"grad_norm": 0.10457716882228851,
"learning_rate": 3.903999999999999e-06,
"loss": 7.312602233886719,
"step": 245
},
{
"epoch": 0.05,
"grad_norm": 0.09624126553535461,
"learning_rate": 3.9839999999999995e-06,
"loss": 7.360851287841797,
"step": 250
},
{
"epoch": 0.051,
"grad_norm": 0.08660672605037689,
"learning_rate": 3.99999300106024e-06,
"loss": 7.353035736083984,
"step": 255
},
{
"epoch": 0.052,
"grad_norm": 0.08430126309394836,
"learning_rate": 3.9999645679514235e-06,
"loss": 7.384159088134766,
"step": 260
},
{
"epoch": 0.053,
"grad_norm": 0.07826503366231918,
"learning_rate": 3.999914263550513e-06,
"loss": 7.4195198059082035,
"step": 265
},
{
"epoch": 0.054,
"grad_norm": 0.08577796816825867,
"learning_rate": 3.9998420884076325e-06,
"loss": 7.362513732910156,
"step": 270
},
{
"epoch": 0.055,
"grad_norm": 0.0649741068482399,
"learning_rate": 3.999748043312075e-06,
"loss": 7.275961303710938,
"step": 275
},
{
"epoch": 0.056,
"grad_norm": 0.07203389704227448,
"learning_rate": 3.999632129292304e-06,
"loss": 7.263797760009766,
"step": 280
},
{
"epoch": 0.057,
"grad_norm": 0.12281067669391632,
"learning_rate": 3.9994943476159364e-06,
"loss": 7.287098693847656,
"step": 285
},
{
"epoch": 0.058,
"grad_norm": 0.07184985280036926,
"learning_rate": 3.999334699789731e-06,
"loss": 7.240232849121094,
"step": 290
},
{
"epoch": 0.059,
"grad_norm": 0.07799801975488663,
"learning_rate": 3.99915318755957e-06,
"loss": 7.2740531921386715,
"step": 295
},
{
"epoch": 0.06,
"grad_norm": 0.0761650875210762,
"learning_rate": 3.9989498129104425e-06,
"loss": 7.299461364746094,
"step": 300
},
{
"epoch": 0.06,
"eval_accuracy": 0.12382905982905983,
"eval_loss": 7.217157363891602,
"eval_runtime": 10.4259,
"eval_samples_per_second": 9.591,
"eval_steps_per_second": 1.631,
"step": 300
}
],
"logging_steps": 5,
"max_steps": 5000,
"num_input_tokens_seen": 0,
"num_train_epochs": 9223372036854775807,
"save_steps": 300,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 6,
"trial_name": null,
"trial_params": null
}