| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.06, |
| "eval_steps": 150, |
| "global_step": 300, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.001, |
| "grad_norm": 0.541594386100769, |
| "learning_rate": 6.4000000000000006e-06, |
| "loss": 10.997595977783202, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.002, |
| "grad_norm": 0.550841212272644, |
| "learning_rate": 1.44e-05, |
| "loss": 10.93243865966797, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.003, |
| "grad_norm": 0.5471506714820862, |
| "learning_rate": 2.2400000000000002e-05, |
| "loss": 10.8595703125, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.004, |
| "grad_norm": 0.5520158410072327, |
| "learning_rate": 3.04e-05, |
| "loss": 10.787896728515625, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.005, |
| "grad_norm": 0.6384730339050293, |
| "learning_rate": 3.8400000000000005e-05, |
| "loss": 10.68493423461914, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.006, |
| "grad_norm": 0.6661590337753296, |
| "learning_rate": 4.64e-05, |
| "loss": 10.593788146972656, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.007, |
| "grad_norm": 0.7120365500450134, |
| "learning_rate": 5.440000000000001e-05, |
| "loss": 10.463143157958985, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.008, |
| "grad_norm": 0.6313916444778442, |
| "learning_rate": 6.24e-05, |
| "loss": 10.36687774658203, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.009, |
| "grad_norm": 0.6490567922592163, |
| "learning_rate": 7.04e-05, |
| "loss": 10.178239440917968, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.01, |
| "grad_norm": 0.6117688417434692, |
| "learning_rate": 7.840000000000001e-05, |
| "loss": 10.030424499511719, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.011, |
| "grad_norm": 0.6446435451507568, |
| "learning_rate": 8.64e-05, |
| "loss": 9.86950225830078, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.012, |
| "grad_norm": 0.5392746329307556, |
| "learning_rate": 9.44e-05, |
| "loss": 9.688623046875, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.013, |
| "grad_norm": 0.5041754841804504, |
| "learning_rate": 0.00010240000000000001, |
| "loss": 9.514056396484374, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.014, |
| "grad_norm": 0.4725438058376312, |
| "learning_rate": 0.00011040000000000001, |
| "loss": 9.406781005859376, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.015, |
| "grad_norm": 0.42810311913490295, |
| "learning_rate": 0.0001184, |
| "loss": 9.310884857177735, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.016, |
| "grad_norm": 0.41262492537498474, |
| "learning_rate": 0.0001264, |
| "loss": 9.21496810913086, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.017, |
| "grad_norm": 0.424248069524765, |
| "learning_rate": 0.00013440000000000001, |
| "loss": 9.244066619873047, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.018, |
| "grad_norm": 0.4215868413448334, |
| "learning_rate": 0.0001424, |
| "loss": 9.153144073486327, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.019, |
| "grad_norm": 0.4235755503177643, |
| "learning_rate": 0.0001504, |
| "loss": 9.029894256591797, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.02, |
| "grad_norm": 0.47701263427734375, |
| "learning_rate": 0.00015840000000000003, |
| "loss": 8.909495544433593, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.021, |
| "grad_norm": 0.5133949518203735, |
| "learning_rate": 0.0001664, |
| "loss": 8.913885498046875, |
| "step": 105 |
| }, |
| { |
| "epoch": 0.022, |
| "grad_norm": 0.43619266152381897, |
| "learning_rate": 0.0001744, |
| "loss": 8.908492279052734, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.023, |
| "grad_norm": 0.46515771746635437, |
| "learning_rate": 0.00018240000000000002, |
| "loss": 8.857489013671875, |
| "step": 115 |
| }, |
| { |
| "epoch": 0.024, |
| "grad_norm": 0.441704660654068, |
| "learning_rate": 0.0001904, |
| "loss": 8.795072174072265, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.025, |
| "grad_norm": 0.4623314142227173, |
| "learning_rate": 0.0001984, |
| "loss": 8.735992431640625, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.026, |
| "grad_norm": 0.3829725980758667, |
| "learning_rate": 0.0002064, |
| "loss": 8.752758026123047, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.027, |
| "grad_norm": 0.34829118847846985, |
| "learning_rate": 0.00021440000000000003, |
| "loss": 8.698936462402344, |
| "step": 135 |
| }, |
| { |
| "epoch": 0.028, |
| "grad_norm": 0.4280807375907898, |
| "learning_rate": 0.00022240000000000004, |
| "loss": 8.654257202148438, |
| "step": 140 |
| }, |
| { |
| "epoch": 0.029, |
| "grad_norm": 0.3860580325126648, |
| "learning_rate": 0.0002304, |
| "loss": 8.588265991210937, |
| "step": 145 |
| }, |
| { |
| "epoch": 0.03, |
| "grad_norm": 0.3843490481376648, |
| "learning_rate": 0.0002384, |
| "loss": 8.613517761230469, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.03, |
| "eval_accuracy": 0.1278876678876679, |
| "eval_loss": 8.568678855895996, |
| "eval_runtime": 10.0729, |
| "eval_samples_per_second": 9.928, |
| "eval_steps_per_second": 1.688, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.031, |
| "grad_norm": 0.3538966774940491, |
| "learning_rate": 0.0002464, |
| "loss": 8.57271957397461, |
| "step": 155 |
| }, |
| { |
| "epoch": 0.032, |
| "grad_norm": 0.4051172435283661, |
| "learning_rate": 0.0002544, |
| "loss": 8.543965911865234, |
| "step": 160 |
| }, |
| { |
| "epoch": 0.033, |
| "grad_norm": 0.35977786779403687, |
| "learning_rate": 0.00026240000000000004, |
| "loss": 8.521247863769531, |
| "step": 165 |
| }, |
| { |
| "epoch": 0.034, |
| "grad_norm": 0.3093051314353943, |
| "learning_rate": 0.0002704, |
| "loss": 8.447119903564452, |
| "step": 170 |
| }, |
| { |
| "epoch": 0.035, |
| "grad_norm": 0.3675788342952728, |
| "learning_rate": 0.0002784, |
| "loss": 8.352106475830078, |
| "step": 175 |
| }, |
| { |
| "epoch": 0.036, |
| "grad_norm": 0.3488416373729706, |
| "learning_rate": 0.0002864, |
| "loss": 8.375956726074218, |
| "step": 180 |
| }, |
| { |
| "epoch": 0.037, |
| "grad_norm": 0.4550418555736542, |
| "learning_rate": 0.0002944, |
| "loss": 8.539746856689453, |
| "step": 185 |
| }, |
| { |
| "epoch": 0.038, |
| "grad_norm": 0.4015609323978424, |
| "learning_rate": 0.00030240000000000003, |
| "loss": 8.388836669921876, |
| "step": 190 |
| }, |
| { |
| "epoch": 0.039, |
| "grad_norm": 0.3297787308692932, |
| "learning_rate": 0.0003104, |
| "loss": 8.376402282714844, |
| "step": 195 |
| }, |
| { |
| "epoch": 0.04, |
| "grad_norm": 0.3928952217102051, |
| "learning_rate": 0.00031840000000000004, |
| "loss": 8.331494140625, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.041, |
| "grad_norm": 0.3978789746761322, |
| "learning_rate": 0.0003264, |
| "loss": 8.267201995849609, |
| "step": 205 |
| }, |
| { |
| "epoch": 0.042, |
| "grad_norm": 0.35222581028938293, |
| "learning_rate": 0.0003344, |
| "loss": 8.33883285522461, |
| "step": 210 |
| }, |
| { |
| "epoch": 0.043, |
| "grad_norm": 0.34856143593788147, |
| "learning_rate": 0.00034240000000000003, |
| "loss": 8.259492492675781, |
| "step": 215 |
| }, |
| { |
| "epoch": 0.044, |
| "grad_norm": 0.38233524560928345, |
| "learning_rate": 0.0003504, |
| "loss": 8.267060852050781, |
| "step": 220 |
| }, |
| { |
| "epoch": 0.045, |
| "grad_norm": 0.3278593420982361, |
| "learning_rate": 0.00035840000000000004, |
| "loss": 8.20487289428711, |
| "step": 225 |
| }, |
| { |
| "epoch": 0.046, |
| "grad_norm": 0.37537282705307007, |
| "learning_rate": 0.0003664, |
| "loss": 8.216254425048827, |
| "step": 230 |
| }, |
| { |
| "epoch": 0.047, |
| "grad_norm": 0.41651761531829834, |
| "learning_rate": 0.00037440000000000005, |
| "loss": 8.180746459960938, |
| "step": 235 |
| }, |
| { |
| "epoch": 0.048, |
| "grad_norm": 0.3697809875011444, |
| "learning_rate": 0.0003824, |
| "loss": 8.194245910644531, |
| "step": 240 |
| }, |
| { |
| "epoch": 0.049, |
| "grad_norm": 0.3911236822605133, |
| "learning_rate": 0.0003904, |
| "loss": 8.119849395751952, |
| "step": 245 |
| }, |
| { |
| "epoch": 0.05, |
| "grad_norm": 0.4369955360889435, |
| "learning_rate": 0.00039840000000000003, |
| "loss": 8.130778503417968, |
| "step": 250 |
| }, |
| { |
| "epoch": 0.051, |
| "grad_norm": 0.34227898716926575, |
| "learning_rate": 0.0003999993001060241, |
| "loss": 8.120016479492188, |
| "step": 255 |
| }, |
| { |
| "epoch": 0.052, |
| "grad_norm": 0.33067235350608826, |
| "learning_rate": 0.00039999645679514236, |
| "loss": 8.143791198730469, |
| "step": 260 |
| }, |
| { |
| "epoch": 0.053, |
| "grad_norm": 0.3268544375896454, |
| "learning_rate": 0.00039999142635505135, |
| "loss": 8.166898345947265, |
| "step": 265 |
| }, |
| { |
| "epoch": 0.054, |
| "grad_norm": 0.5480589866638184, |
| "learning_rate": 0.00039998420884076324, |
| "loss": 8.070804595947266, |
| "step": 270 |
| }, |
| { |
| "epoch": 0.055, |
| "grad_norm": 0.3590603172779083, |
| "learning_rate": 0.00039997480433120753, |
| "loss": 8.069869995117188, |
| "step": 275 |
| }, |
| { |
| "epoch": 0.056, |
| "grad_norm": 0.35152456164360046, |
| "learning_rate": 0.00039996321292923043, |
| "loss": 8.04738540649414, |
| "step": 280 |
| }, |
| { |
| "epoch": 0.057, |
| "grad_norm": 0.3499545156955719, |
| "learning_rate": 0.00039994943476159367, |
| "loss": 8.03879623413086, |
| "step": 285 |
| }, |
| { |
| "epoch": 0.058, |
| "grad_norm": 0.29000213742256165, |
| "learning_rate": 0.00039993346997897316, |
| "loss": 8.007131958007813, |
| "step": 290 |
| }, |
| { |
| "epoch": 0.059, |
| "grad_norm": 0.3708847761154175, |
| "learning_rate": 0.00039991531875595705, |
| "loss": 8.036698913574218, |
| "step": 295 |
| }, |
| { |
| "epoch": 0.06, |
| "grad_norm": 0.3241623640060425, |
| "learning_rate": 0.0003998949812910443, |
| "loss": 8.021849822998046, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.06, |
| "eval_accuracy": 0.1364126984126984, |
| "eval_loss": 8.011240005493164, |
| "eval_runtime": 9.9246, |
| "eval_samples_per_second": 10.076, |
| "eval_steps_per_second": 1.713, |
| "step": 300 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 5000, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 9223372036854775807, |
| "save_steps": 300, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 6, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|