gemma_2_lora_E2b / last-checkpoint /trainer_state.json
HoangVuSnape's picture
Training in progress, step 449, checkpoint
2a6b919 verified
Raw
History Blame Contribute Delete
8.96 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 449,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.022308979364194088,
"grad_norm": 0.47332054376602173,
"learning_rate": 0.0001998790061807298,
"loss": 0.4195059299468994,
"step": 10
},
{
"epoch": 0.044617958728388175,
"grad_norm": 0.6206803917884827,
"learning_rate": 0.00019928708812698545,
"loss": 0.36927917003631594,
"step": 20
},
{
"epoch": 0.06692693809258227,
"grad_norm": 0.5743476748466492,
"learning_rate": 0.00019820494141215104,
"loss": 0.3785541296005249,
"step": 30
},
{
"epoch": 0.08923591745677635,
"grad_norm": 0.49116021394729614,
"learning_rate": 0.00019663790912106393,
"loss": 0.36582486629486083,
"step": 40
},
{
"epoch": 0.11154489682097044,
"grad_norm": 0.6498185992240906,
"learning_rate": 0.00019459372845456705,
"loss": 0.3734700679779053,
"step": 50
},
{
"epoch": 0.13385387618516453,
"grad_norm": 0.6294810175895691,
"learning_rate": 0.0001920824925271838,
"loss": 0.3142746686935425,
"step": 60
},
{
"epoch": 0.1561628555493586,
"grad_norm": 0.610817015171051,
"learning_rate": 0.00018911660053250103,
"loss": 0.3576608657836914,
"step": 70
},
{
"epoch": 0.1784718349135527,
"grad_norm": 0.45995697379112244,
"learning_rate": 0.0001857106965223177,
"loss": 0.30390095710754395,
"step": 80
},
{
"epoch": 0.2007808142777468,
"grad_norm": 1.2754241228103638,
"learning_rate": 0.00018188159710183594,
"loss": 0.3097202301025391,
"step": 90
},
{
"epoch": 0.22308979364194087,
"grad_norm": 0.49120989441871643,
"learning_rate": 0.00017764820839789964,
"loss": 0.30510644912719725,
"step": 100
},
{
"epoch": 0.24539877300613497,
"grad_norm": 0.45583832263946533,
"learning_rate": 0.00017303143271024744,
"loss": 0.2922208786010742,
"step": 110
},
{
"epoch": 0.26770775237032907,
"grad_norm": 0.5388043522834778,
"learning_rate": 0.0001680540653066891,
"loss": 0.2974637508392334,
"step": 120
},
{
"epoch": 0.29001673173452314,
"grad_norm": 0.49362650513648987,
"learning_rate": 0.00016274068187177771,
"loss": 0.2713587760925293,
"step": 130
},
{
"epoch": 0.3123257110987172,
"grad_norm": 1.0019351243972778,
"learning_rate": 0.00015711751716469786,
"loss": 0.392174506187439,
"step": 140
},
{
"epoch": 0.33463469046291133,
"grad_norm": 0.497901052236557,
"learning_rate": 0.0001512123354854955,
"loss": 0.29191443920135496,
"step": 150
},
{
"epoch": 0.3569436698271054,
"grad_norm": 0.6082267761230469,
"learning_rate": 0.00014505429358922,
"loss": 0.3747562408447266,
"step": 160
},
{
"epoch": 0.3792526491912995,
"grad_norm": 0.47960519790649414,
"learning_rate": 0.0001386737967248388,
"loss": 0.3827587842941284,
"step": 170
},
{
"epoch": 0.4015616285554936,
"grad_norm": 0.619045078754425,
"learning_rate": 0.00013210234850972964,
"loss": 0.30683255195617676,
"step": 180
},
{
"epoch": 0.42387060791968767,
"grad_norm": 0.5024956464767456,
"learning_rate": 0.00012537239538099425,
"loss": 0.34638049602508547,
"step": 190
},
{
"epoch": 0.44617958728388174,
"grad_norm": 0.5191444158554077,
"learning_rate": 0.00011851716639161159,
"loss": 0.36590137481689455,
"step": 200
},
{
"epoch": 0.46848856664807587,
"grad_norm": 0.36740896105766296,
"learning_rate": 0.00011157050914243614,
"loss": 0.31072416305541994,
"step": 210
},
{
"epoch": 0.49079754601226994,
"grad_norm": 0.43472257256507874,
"learning_rate": 0.00010456672266012446,
"loss": 0.3114518880844116,
"step": 220
},
{
"epoch": 0.5131065253764641,
"grad_norm": 0.4393475353717804,
"learning_rate": 9.754038804615257e-05,
"loss": 0.2944304943084717,
"step": 230
},
{
"epoch": 0.5354155047406581,
"grad_norm": 0.4425562024116516,
"learning_rate": 9.052619773309317e-05,
"loss": 0.2811075925827026,
"step": 240
},
{
"epoch": 0.5577244841048522,
"grad_norm": 0.4982526898384094,
"learning_rate": 8.355878419119657e-05,
"loss": 0.2992835521697998,
"step": 250
},
{
"epoch": 0.5800334634690463,
"grad_norm": 1.5215120315551758,
"learning_rate": 7.667254893103519e-05,
"loss": 0.3270727634429932,
"step": 260
},
{
"epoch": 0.6023424428332403,
"grad_norm": 0.4536781311035156,
"learning_rate": 6.990149264650814e-05,
"loss": 0.29621281623840334,
"step": 270
},
{
"epoch": 0.6246514221974344,
"grad_norm": 0.47364547848701477,
"learning_rate": 6.32790473368728e-05,
"loss": 0.2747994661331177,
"step": 280
},
{
"epoch": 0.6469604015616286,
"grad_norm": 0.4716944992542267,
"learning_rate": 5.6837911236698536e-05,
"loss": 0.3186234474182129,
"step": 290
},
{
"epoch": 0.6692693809258227,
"grad_norm": 0.42809948325157166,
"learning_rate": 5.060988736877366e-05,
"loss": 0.28460078239440917,
"step": 300
},
{
"epoch": 0.6915783602900167,
"grad_norm": 0.4650896191596985,
"learning_rate": 4.462572651710847e-05,
"loss": 0.30340597629547117,
"step": 310
},
{
"epoch": 0.7138873396542108,
"grad_norm": 0.7915768623352051,
"learning_rate": 3.8914975395353334e-05,
"loss": 0.32752094268798826,
"step": 320
},
{
"epoch": 0.7361963190184049,
"grad_norm": 0.5179622769355774,
"learning_rate": 3.350583076029754e-05,
"loss": 0.3099134206771851,
"step": 330
},
{
"epoch": 0.758505298382599,
"grad_norm": 0.6544927954673767,
"learning_rate": 2.8425000190762353e-05,
"loss": 0.3012081623077393,
"step": 340
},
{
"epoch": 0.7808142777467931,
"grad_norm": 0.4660824239253998,
"learning_rate": 2.3697570219290077e-05,
"loss": 0.26200897693634034,
"step": 350
},
{
"epoch": 0.8031232571109872,
"grad_norm": 0.6966566443443298,
"learning_rate": 1.9346882467727325e-05,
"loss": 0.3476716041564941,
"step": 360
},
{
"epoch": 0.8254322364751813,
"grad_norm": 0.5577861070632935,
"learning_rate": 1.5394418398281352e-05,
"loss": 0.3011465549468994,
"step": 370
},
{
"epoch": 0.8477412158393753,
"grad_norm": 0.559313952922821,
"learning_rate": 1.1859693249089642e-05,
"loss": 0.2599821090698242,
"step": 380
},
{
"epoch": 0.8700501952035694,
"grad_norm": 0.3570460379123688,
"learning_rate": 8.760159677994172e-06,
"loss": 0.2864186763763428,
"step": 390
},
{
"epoch": 0.8923591745677635,
"grad_norm": 0.5391319990158081,
"learning_rate": 6.111121590278346e-06,
"loss": 0.30640947818756104,
"step": 400
},
{
"epoch": 0.9146681539319577,
"grad_norm": 0.3447037637233734,
"learning_rate": 3.925658575840696e-06,
"loss": 0.29050464630126954,
"step": 410
},
{
"epoch": 0.9369771332961517,
"grad_norm": 0.37274837493896484,
"learning_rate": 2.2145613288957478e-06,
"loss": 0.2897591829299927,
"step": 420
},
{
"epoch": 0.9592861126603458,
"grad_norm": 0.43291178345680237,
"learning_rate": 9.862783690666178e-07,
"loss": 0.25869801044464114,
"step": 430
},
{
"epoch": 0.9815950920245399,
"grad_norm": 0.7033930420875549,
"learning_rate": 2.468743269331442e-07,
"loss": 0.2812113046646118,
"step": 440
}
],
"logging_steps": 10,
"max_steps": 449,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 50,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 7296295329092160.0,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}