main-v2-checkpoints / checkpoint-600 /trainer_state.json
leafxyz's picture
Upload folder using huggingface_hub
44f2ec9 verified
Raw
History Blame Contribute Delete
6.93 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.056121971751940884,
"eval_steps": 200,
"global_step": 600,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.0023384154896642037,
"grad_norm": 1.9981695413589478,
"learning_rate": 4.4859813084112145e-06,
"loss": 0.47716766357421875,
"step": 25
},
{
"epoch": 0.004676830979328407,
"grad_norm": 1.9483280181884766,
"learning_rate": 9.158878504672897e-06,
"loss": 0.4592530059814453,
"step": 50
},
{
"epoch": 0.0070152464689926105,
"grad_norm": 2.1290855407714844,
"learning_rate": 1.3831775700934579e-05,
"loss": 0.394501838684082,
"step": 75
},
{
"epoch": 0.009353661958656815,
"grad_norm": 2.8394205570220947,
"learning_rate": 1.8504672897196264e-05,
"loss": 0.3752219009399414,
"step": 100
},
{
"epoch": 0.011692077448321018,
"grad_norm": 2.5833706855773926,
"learning_rate": 2.3177570093457944e-05,
"loss": 0.40475303649902344,
"step": 125
},
{
"epoch": 0.014030492937985221,
"grad_norm": 2.622011661529541,
"learning_rate": 2.7850467289719627e-05,
"loss": 0.45625301361083986,
"step": 150
},
{
"epoch": 0.016368908427649424,
"grad_norm": 2.9406535625457764,
"learning_rate": 3.2523364485981314e-05,
"loss": 0.34918937683105467,
"step": 175
},
{
"epoch": 0.01870732391731363,
"grad_norm": 3.8391785621643066,
"learning_rate": 3.7196261682242994e-05,
"loss": 0.41711517333984377,
"step": 200
},
{
"epoch": 0.01870732391731363,
"eval_pairs_with_negatives_loss": 0.3914637565612793,
"eval_pairs_with_negatives_runtime": 44.0353,
"eval_pairs_with_negatives_samples_per_second": 28.523,
"eval_pairs_with_negatives_steps_per_second": 3.565,
"step": 200
},
{
"epoch": 0.01870732391731363,
"eval_positives_loss": 0.1480681300163269,
"eval_positives_runtime": 51.7193,
"eval_positives_samples_per_second": 42.537,
"eval_positives_steps_per_second": 5.317,
"step": 200
},
{
"epoch": 0.02104573940697783,
"grad_norm": 2.3628170490264893,
"learning_rate": 4.186915887850468e-05,
"loss": 0.4296648406982422,
"step": 225
},
{
"epoch": 0.023384154896642036,
"grad_norm": 2.698835611343384,
"learning_rate": 4.6542056074766354e-05,
"loss": 0.43652393341064455,
"step": 250
},
{
"epoch": 0.02572257038630624,
"grad_norm": 4.1166558265686035,
"learning_rate": 5.121495327102804e-05,
"loss": 0.4344004440307617,
"step": 275
},
{
"epoch": 0.028060985875970442,
"grad_norm": 4.551556587219238,
"learning_rate": 5.588785046728973e-05,
"loss": 0.41842262268066405,
"step": 300
},
{
"epoch": 0.030399401365634647,
"grad_norm": 2.3925657272338867,
"learning_rate": 6.056074766355141e-05,
"loss": 0.4197782897949219,
"step": 325
},
{
"epoch": 0.03273781685529885,
"grad_norm": 3.1023190021514893,
"learning_rate": 6.52336448598131e-05,
"loss": 0.4292975616455078,
"step": 350
},
{
"epoch": 0.03507623234496305,
"grad_norm": 4.1481032371521,
"learning_rate": 6.990654205607478e-05,
"loss": 0.4758753204345703,
"step": 375
},
{
"epoch": 0.03741464783462726,
"grad_norm": 3.8799962997436523,
"learning_rate": 7.457943925233644e-05,
"loss": 0.3311753845214844,
"step": 400
},
{
"epoch": 0.03741464783462726,
"eval_pairs_with_negatives_loss": 0.3694910407066345,
"eval_pairs_with_negatives_runtime": 44.2166,
"eval_pairs_with_negatives_samples_per_second": 28.406,
"eval_pairs_with_negatives_steps_per_second": 3.551,
"step": 400
},
{
"epoch": 0.03741464783462726,
"eval_positives_loss": 0.11797073483467102,
"eval_positives_runtime": 51.3355,
"eval_positives_samples_per_second": 42.855,
"eval_positives_steps_per_second": 5.357,
"step": 400
},
{
"epoch": 0.03975306332429146,
"grad_norm": 3.7068564891815186,
"learning_rate": 7.925233644859813e-05,
"loss": 0.388726806640625,
"step": 425
},
{
"epoch": 0.04209147881395566,
"grad_norm": 2.3407328128814697,
"learning_rate": 8.392523364485981e-05,
"loss": 0.4401784133911133,
"step": 450
},
{
"epoch": 0.04442989430361987,
"grad_norm": 3.0923948287963867,
"learning_rate": 8.85981308411215e-05,
"loss": 0.41050430297851564,
"step": 475
},
{
"epoch": 0.04676830979328407,
"grad_norm": 2.626753568649292,
"learning_rate": 9.327102803738317e-05,
"loss": 0.3923164749145508,
"step": 500
},
{
"epoch": 0.04910672528294827,
"grad_norm": 2.729557991027832,
"learning_rate": 9.794392523364486e-05,
"loss": 0.3163204765319824,
"step": 525
},
{
"epoch": 0.05144514077261248,
"grad_norm": 3.1612608432769775,
"learning_rate": 9.986215045293424e-05,
"loss": 0.35652976989746094,
"step": 550
},
{
"epoch": 0.05378355626227668,
"grad_norm": 2.224350690841675,
"learning_rate": 9.961599054745964e-05,
"loss": 0.37072269439697264,
"step": 575
},
{
"epoch": 0.056121971751940884,
"grad_norm": 3.207382917404175,
"learning_rate": 9.936983064198503e-05,
"loss": 0.3007513427734375,
"step": 600
},
{
"epoch": 0.056121971751940884,
"eval_pairs_with_negatives_loss": 0.3388254940509796,
"eval_pairs_with_negatives_runtime": 43.9558,
"eval_pairs_with_negatives_samples_per_second": 28.574,
"eval_pairs_with_negatives_steps_per_second": 3.572,
"step": 600
},
{
"epoch": 0.056121971751940884,
"eval_positives_loss": 0.10862970352172852,
"eval_positives_runtime": 51.7355,
"eval_positives_samples_per_second": 42.524,
"eval_positives_steps_per_second": 5.315,
"step": 600
}
],
"logging_steps": 25,
"max_steps": 10691,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 200,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 0.0,
"train_batch_size": 32,
"trial_name": null,
"trial_params": null
}