gemma_4_lora_E4b / checkpoint-300 /trainer_state.json
HoangVuSnape's picture
Training in progress, step 300, checkpoint
afc4a4c verified
Raw
History Blame Contribute Delete
7.01 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.6692693809258227,
"eval_steps": 100,
"global_step": 300,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.022308979364194088,
"grad_norm": 4.709710597991943,
"learning_rate": 0.00019999446787180336,
"loss": 3.676974868774414,
"step": 10
},
{
"epoch": 0.044617958728388175,
"grad_norm": 1.4567488431930542,
"learning_rate": 0.00019989613595281384,
"loss": 0.9999975204467774,
"step": 20
},
{
"epoch": 0.06692693809258227,
"grad_norm": 1.1962672472000122,
"learning_rate": 0.00019967500698688952,
"loss": 0.6824670314788819,
"step": 30
},
{
"epoch": 0.08923591745677635,
"grad_norm": 0.8628137707710266,
"learning_rate": 0.0001993313527961959,
"loss": 0.5993218421936035,
"step": 40
},
{
"epoch": 0.11154489682097044,
"grad_norm": 1.0308934450149536,
"learning_rate": 0.00019886559581668999,
"loss": 0.542162561416626,
"step": 50
},
{
"epoch": 0.13385387618516453,
"grad_norm": 0.9261613488197327,
"learning_rate": 0.00019827830857884173,
"loss": 0.44734554290771483,
"step": 60
},
{
"epoch": 0.1561628555493586,
"grad_norm": 0.9318423867225647,
"learning_rate": 0.00019757021300385286,
"loss": 0.44412760734558104,
"step": 70
},
{
"epoch": 0.1784718349135527,
"grad_norm": 0.5784896612167358,
"learning_rate": 0.00019674217951623707,
"loss": 0.40378632545471194,
"step": 80
},
{
"epoch": 0.2007808142777468,
"grad_norm": 0.7836484909057617,
"learning_rate": 0.00019579522597385315,
"loss": 0.3856403350830078,
"step": 90
},
{
"epoch": 0.22308979364194087,
"grad_norm": 0.6503493189811707,
"learning_rate": 0.00019473051641670606,
"loss": 0.3800614356994629,
"step": 100
},
{
"epoch": 0.22308979364194087,
"eval_loss": 8.158984184265137,
"eval_runtime": 248.821,
"eval_samples_per_second": 1.813,
"eval_steps_per_second": 1.813,
"step": 100
},
{
"epoch": 0.24539877300613497,
"grad_norm": 0.582433819770813,
"learning_rate": 0.00019354935963605393,
"loss": 0.34090685844421387,
"step": 110
},
{
"epoch": 0.26770775237032907,
"grad_norm": 0.6219270825386047,
"learning_rate": 0.00019225320756558023,
"loss": 0.3560009956359863,
"step": 120
},
{
"epoch": 0.29001673173452314,
"grad_norm": 0.5081164836883545,
"learning_rate": 0.0001908436534966081,
"loss": 0.32209837436676025,
"step": 130
},
{
"epoch": 0.3123257110987172,
"grad_norm": 1.0672804117202759,
"learning_rate": 0.00018932243011955154,
"loss": 0.4280549049377441,
"step": 140
},
{
"epoch": 0.33463469046291133,
"grad_norm": 0.5716775059700012,
"learning_rate": 0.00018769140739401062,
"loss": 0.29103860855102537,
"step": 150
},
{
"epoch": 0.3569436698271054,
"grad_norm": 0.6793851256370544,
"learning_rate": 0.00018595259025012877,
"loss": 0.3713259696960449,
"step": 160
},
{
"epoch": 0.3792526491912995,
"grad_norm": 0.6091159582138062,
"learning_rate": 0.0001841081161240379,
"loss": 0.3802988529205322,
"step": 170
},
{
"epoch": 0.4015616285554936,
"grad_norm": 0.7563126087188721,
"learning_rate": 0.0001821602523304211,
"loss": 0.3028958082199097,
"step": 180
},
{
"epoch": 0.42387060791968767,
"grad_norm": 0.5827437043190002,
"learning_rate": 0.00018011139327542237,
"loss": 0.3516202211380005,
"step": 190
},
{
"epoch": 0.44617958728388174,
"grad_norm": 0.6265884041786194,
"learning_rate": 0.0001779640575133296,
"loss": 0.3505818843841553,
"step": 200
},
{
"epoch": 0.44617958728388174,
"eval_loss": 6.77685546875,
"eval_runtime": 246.6844,
"eval_samples_per_second": 1.828,
"eval_steps_per_second": 1.828,
"step": 200
},
{
"epoch": 0.46848856664807587,
"grad_norm": 0.4515160024166107,
"learning_rate": 0.00017572088465064848,
"loss": 0.30297350883483887,
"step": 210
},
{
"epoch": 0.49079754601226994,
"grad_norm": 0.47408005595207214,
"learning_rate": 0.0001733846321013738,
"loss": 0.30213174819946287,
"step": 220
},
{
"epoch": 0.5131065253764641,
"grad_norm": 0.43776413798332214,
"learning_rate": 0.00017095817169744595,
"loss": 0.2981758117675781,
"step": 230
},
{
"epoch": 0.5354155047406581,
"grad_norm": 0.5996765494346619,
"learning_rate": 0.00016844448615855933,
"loss": 0.2770395278930664,
"step": 240
},
{
"epoch": 0.5577244841048522,
"grad_norm": 1.4966278076171875,
"learning_rate": 0.0001658466654256627,
"loss": 0.3032949924468994,
"step": 250
},
{
"epoch": 0.5800334634690463,
"grad_norm": 0.5301216840744019,
"learning_rate": 0.00016316790286265763,
"loss": 0.3215930700302124,
"step": 260
},
{
"epoch": 0.6023424428332403,
"grad_norm": 0.5374172925949097,
"learning_rate": 0.00016041149133096512,
"loss": 0.29086947441101074,
"step": 270
},
{
"epoch": 0.6246514221974344,
"grad_norm": 0.5407304167747498,
"learning_rate": 0.00015758081914178456,
"loss": 0.26963791847229,
"step": 280
},
{
"epoch": 0.6469604015616286,
"grad_norm": 0.40757226943969727,
"learning_rate": 0.00015467936589102176,
"loss": 0.3320644378662109,
"step": 290
},
{
"epoch": 0.6692693809258227,
"grad_norm": 0.36359116435050964,
"learning_rate": 0.00015171069818200548,
"loss": 0.28259494304656985,
"step": 300
},
{
"epoch": 0.6692693809258227,
"eval_loss": 6.212057590484619,
"eval_runtime": 246.7095,
"eval_samples_per_second": 1.828,
"eval_steps_per_second": 1.828,
"step": 300
}
],
"logging_steps": 10,
"max_steps": 898,
"num_input_tokens_seen": 0,
"num_train_epochs": 2,
"save_steps": 50,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 9148716388151424.0,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}