hadxs's picture
Upload folder using huggingface_hub
9c3f8b7 verified
Raw
History Blame Contribute Delete
5.21 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.00461968447555032,
"eval_steps": 500,
"global_step": 500,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.0001847873790220128,
"grad_norm": 4.357386112213135,
"learning_rate": 6.4e-07,
"loss": 2.7993131637573243,
"step": 20
},
{
"epoch": 0.0003695747580440256,
"grad_norm": 4.551547527313232,
"learning_rate": 1.44e-06,
"loss": 2.639370155334473,
"step": 40
},
{
"epoch": 0.0005543621370660384,
"grad_norm": 4.46128511428833,
"learning_rate": 2.24e-06,
"loss": 2.696819877624512,
"step": 60
},
{
"epoch": 0.0007391495160880512,
"grad_norm": 1.5103334188461304,
"learning_rate": 3.04e-06,
"loss": 2.2109146118164062,
"step": 80
},
{
"epoch": 0.000923936895110064,
"grad_norm": 1.0658206939697266,
"learning_rate": 3.84e-06,
"loss": 2.2014698028564452,
"step": 100
},
{
"epoch": 0.0011087242741320768,
"grad_norm": 1.4809857606887817,
"learning_rate": 4.64e-06,
"loss": 1.9168745040893556,
"step": 120
},
{
"epoch": 0.0012935116531540896,
"grad_norm": 1.5375170707702637,
"learning_rate": 5.44e-06,
"loss": 1.8440933227539062,
"step": 140
},
{
"epoch": 0.0014782990321761025,
"grad_norm": 0.745366096496582,
"learning_rate": 6.24e-06,
"loss": 1.7943439483642578,
"step": 160
},
{
"epoch": 0.001663086411198115,
"grad_norm": 0.5929296612739563,
"learning_rate": 7.04e-06,
"loss": 1.6178794860839845,
"step": 180
},
{
"epoch": 0.001847873790220128,
"grad_norm": 0.8178684115409851,
"learning_rate": 7.8e-06,
"loss": 1.495293140411377,
"step": 200
},
{
"epoch": 0.002032661169242141,
"grad_norm": 1.090906023979187,
"learning_rate": 8.599999999999999e-06,
"loss": 1.5392029762268067,
"step": 220
},
{
"epoch": 0.0022174485482641536,
"grad_norm": 0.7013095021247864,
"learning_rate": 9.4e-06,
"loss": 1.4353426933288573,
"step": 240
},
{
"epoch": 0.002402235927286166,
"grad_norm": 0.8774408102035522,
"learning_rate": 1.02e-05,
"loss": 1.508979892730713,
"step": 260
},
{
"epoch": 0.0025870233063081793,
"grad_norm": 0.6000725626945496,
"learning_rate": 1.1000000000000001e-05,
"loss": 1.4708028793334962,
"step": 280
},
{
"epoch": 0.002771810685330192,
"grad_norm": 2.499645471572876,
"learning_rate": 1.18e-05,
"loss": 1.4927824020385743,
"step": 300
},
{
"epoch": 0.002956598064352205,
"grad_norm": 0.46313002705574036,
"learning_rate": 1.2600000000000001e-05,
"loss": 1.4648756980895996,
"step": 320
},
{
"epoch": 0.0031413854433742176,
"grad_norm": 0.44169697165489197,
"learning_rate": 1.3400000000000002e-05,
"loss": 1.3465867042541504,
"step": 340
},
{
"epoch": 0.00332617282239623,
"grad_norm": 0.6739365458488464,
"learning_rate": 1.42e-05,
"loss": 1.425154209136963,
"step": 360
},
{
"epoch": 0.0035109602014182432,
"grad_norm": 1.1867340803146362,
"learning_rate": 1.5e-05,
"loss": 1.4140001296997071,
"step": 380
},
{
"epoch": 0.003695747580440256,
"grad_norm": 0.5072050094604492,
"learning_rate": 1.58e-05,
"loss": 1.3576015472412108,
"step": 400
},
{
"epoch": 0.003880534959462269,
"grad_norm": 0.7821234464645386,
"learning_rate": 1.66e-05,
"loss": 1.3783547401428222,
"step": 420
},
{
"epoch": 0.004065322338484282,
"grad_norm": 0.547825813293457,
"learning_rate": 1.74e-05,
"loss": 1.3415006637573241,
"step": 440
},
{
"epoch": 0.004250109717506294,
"grad_norm": 0.5836990475654602,
"learning_rate": 1.8200000000000002e-05,
"loss": 1.3484737396240234,
"step": 460
},
{
"epoch": 0.004434897096528307,
"grad_norm": 0.4791603982448578,
"learning_rate": 1.9e-05,
"loss": 1.3799403190612793,
"step": 480
},
{
"epoch": 0.00461968447555032,
"grad_norm": 1.0716239213943481,
"learning_rate": 1.9800000000000004e-05,
"loss": 1.3633416175842286,
"step": 500
}
],
"logging_steps": 20,
"max_steps": 100000,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 3.004783929950208e+16,
"train_batch_size": 2,
"trial_name": null,
"trial_params": null
}