| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.00461968447555032, |
| "eval_steps": 500, |
| "global_step": 500, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.0001847873790220128, |
| "grad_norm": 4.357386112213135, |
| "learning_rate": 6.4e-07, |
| "loss": 2.7993131637573243, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.0003695747580440256, |
| "grad_norm": 4.551547527313232, |
| "learning_rate": 1.44e-06, |
| "loss": 2.639370155334473, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.0005543621370660384, |
| "grad_norm": 4.46128511428833, |
| "learning_rate": 2.24e-06, |
| "loss": 2.696819877624512, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.0007391495160880512, |
| "grad_norm": 1.5103334188461304, |
| "learning_rate": 3.04e-06, |
| "loss": 2.2109146118164062, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.000923936895110064, |
| "grad_norm": 1.0658206939697266, |
| "learning_rate": 3.84e-06, |
| "loss": 2.2014698028564452, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.0011087242741320768, |
| "grad_norm": 1.4809857606887817, |
| "learning_rate": 4.64e-06, |
| "loss": 1.9168745040893556, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.0012935116531540896, |
| "grad_norm": 1.5375170707702637, |
| "learning_rate": 5.44e-06, |
| "loss": 1.8440933227539062, |
| "step": 140 |
| }, |
| { |
| "epoch": 0.0014782990321761025, |
| "grad_norm": 0.745366096496582, |
| "learning_rate": 6.24e-06, |
| "loss": 1.7943439483642578, |
| "step": 160 |
| }, |
| { |
| "epoch": 0.001663086411198115, |
| "grad_norm": 0.5929296612739563, |
| "learning_rate": 7.04e-06, |
| "loss": 1.6178794860839845, |
| "step": 180 |
| }, |
| { |
| "epoch": 0.001847873790220128, |
| "grad_norm": 0.8178684115409851, |
| "learning_rate": 7.8e-06, |
| "loss": 1.495293140411377, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.002032661169242141, |
| "grad_norm": 1.090906023979187, |
| "learning_rate": 8.599999999999999e-06, |
| "loss": 1.5392029762268067, |
| "step": 220 |
| }, |
| { |
| "epoch": 0.0022174485482641536, |
| "grad_norm": 0.7013095021247864, |
| "learning_rate": 9.4e-06, |
| "loss": 1.4353426933288573, |
| "step": 240 |
| }, |
| { |
| "epoch": 0.002402235927286166, |
| "grad_norm": 0.8774408102035522, |
| "learning_rate": 1.02e-05, |
| "loss": 1.508979892730713, |
| "step": 260 |
| }, |
| { |
| "epoch": 0.0025870233063081793, |
| "grad_norm": 0.6000725626945496, |
| "learning_rate": 1.1000000000000001e-05, |
| "loss": 1.4708028793334962, |
| "step": 280 |
| }, |
| { |
| "epoch": 0.002771810685330192, |
| "grad_norm": 2.499645471572876, |
| "learning_rate": 1.18e-05, |
| "loss": 1.4927824020385743, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.002956598064352205, |
| "grad_norm": 0.46313002705574036, |
| "learning_rate": 1.2600000000000001e-05, |
| "loss": 1.4648756980895996, |
| "step": 320 |
| }, |
| { |
| "epoch": 0.0031413854433742176, |
| "grad_norm": 0.44169697165489197, |
| "learning_rate": 1.3400000000000002e-05, |
| "loss": 1.3465867042541504, |
| "step": 340 |
| }, |
| { |
| "epoch": 0.00332617282239623, |
| "grad_norm": 0.6739365458488464, |
| "learning_rate": 1.42e-05, |
| "loss": 1.425154209136963, |
| "step": 360 |
| }, |
| { |
| "epoch": 0.0035109602014182432, |
| "grad_norm": 1.1867340803146362, |
| "learning_rate": 1.5e-05, |
| "loss": 1.4140001296997071, |
| "step": 380 |
| }, |
| { |
| "epoch": 0.003695747580440256, |
| "grad_norm": 0.5072050094604492, |
| "learning_rate": 1.58e-05, |
| "loss": 1.3576015472412108, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.003880534959462269, |
| "grad_norm": 0.7821234464645386, |
| "learning_rate": 1.66e-05, |
| "loss": 1.3783547401428222, |
| "step": 420 |
| }, |
| { |
| "epoch": 0.004065322338484282, |
| "grad_norm": 0.547825813293457, |
| "learning_rate": 1.74e-05, |
| "loss": 1.3415006637573241, |
| "step": 440 |
| }, |
| { |
| "epoch": 0.004250109717506294, |
| "grad_norm": 0.5836990475654602, |
| "learning_rate": 1.8200000000000002e-05, |
| "loss": 1.3484737396240234, |
| "step": 460 |
| }, |
| { |
| "epoch": 0.004434897096528307, |
| "grad_norm": 0.4791603982448578, |
| "learning_rate": 1.9e-05, |
| "loss": 1.3799403190612793, |
| "step": 480 |
| }, |
| { |
| "epoch": 0.00461968447555032, |
| "grad_norm": 1.0716239213943481, |
| "learning_rate": 1.9800000000000004e-05, |
| "loss": 1.3633416175842286, |
| "step": 500 |
| } |
| ], |
| "logging_steps": 20, |
| "max_steps": 100000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 1, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 3.004783929950208e+16, |
| "train_batch_size": 2, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|