| { |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.6811409110259685, |
| "eval_steps": 500, |
| "global_step": 200, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.017028522775649212, |
| "grad_norm": 3.096078395843506, |
| "learning_rate": 0.0001988623435722412, |
| "loss": 2.9252, |
| "mean_token_accuracy": 0.5491172857582569, |
| "num_tokens": 4766.0, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.034057045551298425, |
| "grad_norm": 0.695277214050293, |
| "learning_rate": 0.00019772468714448238, |
| "loss": 0.924, |
| "mean_token_accuracy": 0.8361630752682686, |
| "num_tokens": 9556.0, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.05108556832694764, |
| "grad_norm": 0.541026771068573, |
| "learning_rate": 0.00019658703071672356, |
| "loss": 0.5477, |
| "mean_token_accuracy": 0.8945453137159347, |
| "num_tokens": 14116.0, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.06811409110259685, |
| "grad_norm": 0.5480035543441772, |
| "learning_rate": 0.00019544937428896475, |
| "loss": 0.5796, |
| "mean_token_accuracy": 0.8836526513099671, |
| "num_tokens": 18879.0, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.08514261387824607, |
| "grad_norm": 0.5408394932746887, |
| "learning_rate": 0.00019431171786120593, |
| "loss": 0.6289, |
| "mean_token_accuracy": 0.8782178282737731, |
| "num_tokens": 23699.0, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.10217113665389528, |
| "grad_norm": 0.583755612373352, |
| "learning_rate": 0.00019317406143344712, |
| "loss": 0.5507, |
| "mean_token_accuracy": 0.8905435591936112, |
| "num_tokens": 28425.0, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.11919965942954448, |
| "grad_norm": 0.5654754638671875, |
| "learning_rate": 0.0001920364050056883, |
| "loss": 0.5163, |
| "mean_token_accuracy": 0.8963450387120246, |
| "num_tokens": 33060.0, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.1362281822051937, |
| "grad_norm": 0.5462912321090698, |
| "learning_rate": 0.00019089874857792946, |
| "loss": 0.6819, |
| "mean_token_accuracy": 0.8708900690078736, |
| "num_tokens": 38017.0, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.1532567049808429, |
| "grad_norm": 0.45884689688682556, |
| "learning_rate": 0.00018976109215017067, |
| "loss": 0.5452, |
| "mean_token_accuracy": 0.8896441653370857, |
| "num_tokens": 42760.0, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.17028522775649213, |
| "grad_norm": 0.5886663198471069, |
| "learning_rate": 0.00018862343572241183, |
| "loss": 0.5337, |
| "mean_token_accuracy": 0.8954117730259895, |
| "num_tokens": 47455.0, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.18731375053214133, |
| "grad_norm": 0.5249215364456177, |
| "learning_rate": 0.00018748577929465302, |
| "loss": 0.4948, |
| "mean_token_accuracy": 0.9001417383551598, |
| "num_tokens": 52186.0, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.20434227330779056, |
| "grad_norm": 0.40858832001686096, |
| "learning_rate": 0.0001863481228668942, |
| "loss": 0.4616, |
| "mean_token_accuracy": 0.9116851478815079, |
| "num_tokens": 56765.0, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.22137079608343976, |
| "grad_norm": 0.4287504553794861, |
| "learning_rate": 0.0001852104664391354, |
| "loss": 0.5232, |
| "mean_token_accuracy": 0.895601749420166, |
| "num_tokens": 61436.0, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.23839931885908897, |
| "grad_norm": 0.5480343103408813, |
| "learning_rate": 0.00018407281001137657, |
| "loss": 0.5757, |
| "mean_token_accuracy": 0.8878508538007737, |
| "num_tokens": 66190.0, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.2554278416347382, |
| "grad_norm": 0.4539305567741394, |
| "learning_rate": 0.00018293515358361776, |
| "loss": 0.5136, |
| "mean_token_accuracy": 0.8946248903870583, |
| "num_tokens": 70888.0, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.2724563644103874, |
| "grad_norm": 0.5497062802314758, |
| "learning_rate": 0.00018179749715585894, |
| "loss": 0.4996, |
| "mean_token_accuracy": 0.9008801370859146, |
| "num_tokens": 75588.0, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.2894848871860366, |
| "grad_norm": 0.5979815721511841, |
| "learning_rate": 0.00018065984072810013, |
| "loss": 0.5454, |
| "mean_token_accuracy": 0.8904410973191261, |
| "num_tokens": 80394.0, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.3065134099616858, |
| "grad_norm": 0.44527795910835266, |
| "learning_rate": 0.0001795221843003413, |
| "loss": 0.4675, |
| "mean_token_accuracy": 0.9040657833218575, |
| "num_tokens": 85088.0, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.32354193273733506, |
| "grad_norm": 0.5105463862419128, |
| "learning_rate": 0.0001783845278725825, |
| "loss": 0.5391, |
| "mean_token_accuracy": 0.8978576123714447, |
| "num_tokens": 89799.0, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.34057045551298426, |
| "grad_norm": 0.5285545587539673, |
| "learning_rate": 0.00017724687144482368, |
| "loss": 0.5426, |
| "mean_token_accuracy": 0.890034094452858, |
| "num_tokens": 94573.0, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.35759897828863346, |
| "grad_norm": 0.5576184988021851, |
| "learning_rate": 0.00017610921501706487, |
| "loss": 0.5258, |
| "mean_token_accuracy": 0.8902270376682282, |
| "num_tokens": 99357.0, |
| "step": 105 |
| }, |
| { |
| "epoch": 0.37462750106428266, |
| "grad_norm": 0.5248317718505859, |
| "learning_rate": 0.00017497155858930602, |
| "loss": 0.5373, |
| "mean_token_accuracy": 0.8933822825551033, |
| "num_tokens": 104092.0, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.39165602383993187, |
| "grad_norm": 0.45350518822669983, |
| "learning_rate": 0.00017383390216154724, |
| "loss": 0.5542, |
| "mean_token_accuracy": 0.8876873329281807, |
| "num_tokens": 108902.0, |
| "step": 115 |
| }, |
| { |
| "epoch": 0.4086845466155811, |
| "grad_norm": 0.5155291557312012, |
| "learning_rate": 0.0001726962457337884, |
| "loss": 0.5019, |
| "mean_token_accuracy": 0.9029773235321045, |
| "num_tokens": 113571.0, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.4257130693912303, |
| "grad_norm": 0.5053763389587402, |
| "learning_rate": 0.0001715585893060296, |
| "loss": 0.5595, |
| "mean_token_accuracy": 0.8906429842114448, |
| "num_tokens": 118402.0, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.4427415921668795, |
| "grad_norm": 0.5493877530097961, |
| "learning_rate": 0.00017042093287827076, |
| "loss": 0.4821, |
| "mean_token_accuracy": 0.9028852418065071, |
| "num_tokens": 123147.0, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.45977011494252873, |
| "grad_norm": 0.426513135433197, |
| "learning_rate": 0.00016928327645051198, |
| "loss": 0.5311, |
| "mean_token_accuracy": 0.8920483842492104, |
| "num_tokens": 127878.0, |
| "step": 135 |
| }, |
| { |
| "epoch": 0.47679863771817793, |
| "grad_norm": 0.5025385022163391, |
| "learning_rate": 0.00016814562002275313, |
| "loss": 0.4699, |
| "mean_token_accuracy": 0.9039205580949783, |
| "num_tokens": 132591.0, |
| "step": 140 |
| }, |
| { |
| "epoch": 0.49382716049382713, |
| "grad_norm": 0.4677104651927948, |
| "learning_rate": 0.00016700796359499432, |
| "loss": 0.4876, |
| "mean_token_accuracy": 0.9038951709866524, |
| "num_tokens": 137221.0, |
| "step": 145 |
| }, |
| { |
| "epoch": 0.5108556832694764, |
| "grad_norm": 0.5534231066703796, |
| "learning_rate": 0.0001658703071672355, |
| "loss": 0.4969, |
| "mean_token_accuracy": 0.897729343175888, |
| "num_tokens": 141981.0, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.5278842060451255, |
| "grad_norm": 0.4998553395271301, |
| "learning_rate": 0.0001647326507394767, |
| "loss": 0.5476, |
| "mean_token_accuracy": 0.8921540513634681, |
| "num_tokens": 146807.0, |
| "step": 155 |
| }, |
| { |
| "epoch": 0.5449127288207748, |
| "grad_norm": 0.545978307723999, |
| "learning_rate": 0.00016359499431171787, |
| "loss": 0.4543, |
| "mean_token_accuracy": 0.9018902003765106, |
| "num_tokens": 151544.0, |
| "step": 160 |
| }, |
| { |
| "epoch": 0.561941251596424, |
| "grad_norm": 0.5749042630195618, |
| "learning_rate": 0.00016245733788395906, |
| "loss": 0.5064, |
| "mean_token_accuracy": 0.8976296812295914, |
| "num_tokens": 156280.0, |
| "step": 165 |
| }, |
| { |
| "epoch": 0.5789697743720732, |
| "grad_norm": 0.5217350125312805, |
| "learning_rate": 0.00016131968145620024, |
| "loss": 0.4741, |
| "mean_token_accuracy": 0.9045222416520119, |
| "num_tokens": 161011.0, |
| "step": 170 |
| }, |
| { |
| "epoch": 0.5959982971477225, |
| "grad_norm": 0.5088939666748047, |
| "learning_rate": 0.00016018202502844143, |
| "loss": 0.4655, |
| "mean_token_accuracy": 0.9078934848308563, |
| "num_tokens": 165670.0, |
| "step": 175 |
| }, |
| { |
| "epoch": 0.6130268199233716, |
| "grad_norm": 0.5888277292251587, |
| "learning_rate": 0.0001590443686006826, |
| "loss": 0.4692, |
| "mean_token_accuracy": 0.9044368535280227, |
| "num_tokens": 170377.0, |
| "step": 180 |
| }, |
| { |
| "epoch": 0.6300553426990209, |
| "grad_norm": 0.6187074780464172, |
| "learning_rate": 0.0001579067121729238, |
| "loss": 0.5105, |
| "mean_token_accuracy": 0.8983473464846611, |
| "num_tokens": 175090.0, |
| "step": 185 |
| }, |
| { |
| "epoch": 0.6470838654746701, |
| "grad_norm": 0.5764688849449158, |
| "learning_rate": 0.00015676905574516496, |
| "loss": 0.5022, |
| "mean_token_accuracy": 0.8986694201827049, |
| "num_tokens": 179933.0, |
| "step": 190 |
| }, |
| { |
| "epoch": 0.6641123882503193, |
| "grad_norm": 0.5491649508476257, |
| "learning_rate": 0.00015563139931740617, |
| "loss": 0.5028, |
| "mean_token_accuracy": 0.8957466080784797, |
| "num_tokens": 184767.0, |
| "step": 195 |
| }, |
| { |
| "epoch": 0.6811409110259685, |
| "grad_norm": 0.4836632311344147, |
| "learning_rate": 0.00015449374288964733, |
| "loss": 0.5325, |
| "mean_token_accuracy": 0.8894904315471649, |
| "num_tokens": 189624.0, |
| "step": 200 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 879, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 3, |
| "save_steps": 200, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 8090485423792128.0, |
| "train_batch_size": 1, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|