| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.08, |
| "eval_steps": 1000, |
| "global_step": 4000, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": false, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.002, |
| "grad_norm": 19.660429000854492, |
| "learning_rate": 7.920000000000001e-05, |
| "loss": 67.8543, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 24.802082061767578, |
| "learning_rate": 0.00015920000000000002, |
| "loss": 54.1444, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 16.059066772460938, |
| "learning_rate": 0.00023920000000000001, |
| "loss": 51.5921, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 27.705520629882812, |
| "learning_rate": 0.0003192, |
| "loss": 50.0536, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 25.79052734375, |
| "learning_rate": 0.0003992, |
| "loss": 48.6987, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 14.82126522064209, |
| "learning_rate": 0.00047920000000000005, |
| "loss": 47.3182, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 12.00988483428955, |
| "learning_rate": 0.0005592, |
| "loss": 46.6574, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 32.48664474487305, |
| "learning_rate": 0.0006392, |
| "loss": 46.0747, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 25.646282196044922, |
| "learning_rate": 0.0007191999999999999, |
| "loss": 45.6098, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 16.071203231811523, |
| "learning_rate": 0.0007992, |
| "loss": 44.8537, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.02, |
| "eval_accuracy": 0.18606457925636008, |
| "eval_loss": 44.18140411376953, |
| "eval_runtime": 14.581, |
| "eval_samples_per_second": 8.573, |
| "eval_steps_per_second": 0.137, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 14.277446746826172, |
| "learning_rate": 0.0008792, |
| "loss": 44.223, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 9.809870719909668, |
| "learning_rate": 0.0009592000000000001, |
| "loss": 43.5719, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 16.994014739990234, |
| "learning_rate": 0.0010391999999999999, |
| "loss": 43.0323, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 8.082005500793457, |
| "learning_rate": 0.0011192, |
| "loss": 42.3631, |
| "step": 1400 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 7.080636024475098, |
| "learning_rate": 0.0011992, |
| "loss": 42.0285, |
| "step": 1500 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 8.547815322875977, |
| "learning_rate": 0.0012791999999999999, |
| "loss": 42.6829, |
| "step": 1600 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 7.541939735412598, |
| "learning_rate": 0.0013592, |
| "loss": 41.9796, |
| "step": 1700 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 6.902923583984375, |
| "learning_rate": 0.0014392, |
| "loss": 41.5016, |
| "step": 1800 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 7.68479061126709, |
| "learning_rate": 0.0015192, |
| "loss": 41.2607, |
| "step": 1900 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 8.936702728271484, |
| "learning_rate": 0.0015992, |
| "loss": 40.7545, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.04, |
| "eval_accuracy": 0.21657142857142858, |
| "eval_loss": 40.417686462402344, |
| "eval_runtime": 3.051, |
| "eval_samples_per_second": 40.97, |
| "eval_steps_per_second": 0.656, |
| "step": 2000 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 5.834764003753662, |
| "learning_rate": 0.0016792, |
| "loss": 40.2024, |
| "step": 2100 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 3.9864397048950195, |
| "learning_rate": 0.0017592, |
| "loss": 39.4534, |
| "step": 2200 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 6.052990436553955, |
| "learning_rate": 0.0018392, |
| "loss": 40.0088, |
| "step": 2300 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 10.583518981933594, |
| "learning_rate": 0.0019192, |
| "loss": 40.099, |
| "step": 2400 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 3.686554193496704, |
| "learning_rate": 0.0019992, |
| "loss": 39.5664, |
| "step": 2500 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 2.981701612472534, |
| "learning_rate": 0.001999978563623903, |
| "loss": 39.5087, |
| "step": 2600 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 3.382185459136963, |
| "learning_rate": 0.0019999133871331223, |
| "loss": 38.9185, |
| "step": 2700 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 2.8874025344848633, |
| "learning_rate": 0.0019998044711915177, |
| "loss": 38.1472, |
| "step": 2800 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 3.083789825439453, |
| "learning_rate": 0.0019996518205634257, |
| "loss": 37.5833, |
| "step": 2900 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 2.315279483795166, |
| "learning_rate": 0.0019994554419262797, |
| "loss": 37.927, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_accuracy": 0.23828180039138944, |
| "eval_loss": 37.86294174194336, |
| "eval_runtime": 3.1437, |
| "eval_samples_per_second": 39.763, |
| "eval_steps_per_second": 0.636, |
| "step": 3000 |
| }, |
| { |
| "epoch": 0.062, |
| "grad_norm": 4.6366777420043945, |
| "learning_rate": 0.001999215343870317, |
| "loss": 38.432, |
| "step": 3100 |
| }, |
| { |
| "epoch": 0.064, |
| "grad_norm": 2.8195507526397705, |
| "learning_rate": 0.0019989315368982054, |
| "loss": 37.8144, |
| "step": 3200 |
| }, |
| { |
| "epoch": 0.066, |
| "grad_norm": 2.320726156234741, |
| "learning_rate": 0.0019986040334245797, |
| "loss": 38.1739, |
| "step": 3300 |
| }, |
| { |
| "epoch": 0.068, |
| "grad_norm": 2.575469732284546, |
| "learning_rate": 0.001998232847775504, |
| "loss": 37.1996, |
| "step": 3400 |
| }, |
| { |
| "epoch": 0.07, |
| "grad_norm": 2.4692764282226562, |
| "learning_rate": 0.0019978179961878404, |
| "loss": 36.8725, |
| "step": 3500 |
| }, |
| { |
| "epoch": 0.072, |
| "grad_norm": 2.555267810821533, |
| "learning_rate": 0.001997359496808541, |
| "loss": 36.2038, |
| "step": 3600 |
| }, |
| { |
| "epoch": 0.074, |
| "grad_norm": 1.9989144802093506, |
| "learning_rate": 0.001996857369693855, |
| "loss": 36.3164, |
| "step": 3700 |
| }, |
| { |
| "epoch": 0.076, |
| "grad_norm": 2.223381519317627, |
| "learning_rate": 0.0019963116368084486, |
| "loss": 37.1086, |
| "step": 3800 |
| }, |
| { |
| "epoch": 0.078, |
| "grad_norm": 1.996813416481018, |
| "learning_rate": 0.001995722322024446, |
| "loss": 36.5327, |
| "step": 3900 |
| }, |
| { |
| "epoch": 0.08, |
| "grad_norm": 1.8378113508224487, |
| "learning_rate": 0.001995089451120385, |
| "loss": 36.314, |
| "step": 4000 |
| }, |
| { |
| "epoch": 0.08, |
| "eval_accuracy": 0.2553561643835616, |
| "eval_loss": 36.08281707763672, |
| "eval_runtime": 3.1615, |
| "eval_samples_per_second": 39.538, |
| "eval_steps_per_second": 0.633, |
| "step": 4000 |
| } |
| ], |
| "logging_steps": 100, |
| "max_steps": 50000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 2000, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 8, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|