| { |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.6811409110259685, |
| "eval_steps": 500, |
| "global_step": 200, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.017028522775649212, |
| "grad_norm": 2.3286359310150146, |
| "learning_rate": 0.0001988623435722412, |
| "loss": 4.1641, |
| "mean_token_accuracy": 0.3866006717085838, |
| "num_tokens": 4326.0, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.034057045551298425, |
| "grad_norm": 1.4926602840423584, |
| "learning_rate": 0.00019772468714448238, |
| "loss": 2.0591, |
| "mean_token_accuracy": 0.6790341928601265, |
| "num_tokens": 8676.0, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.05108556832694764, |
| "grad_norm": 1.0835968255996704, |
| "learning_rate": 0.00019658703071672356, |
| "loss": 1.3145, |
| "mean_token_accuracy": 0.8256841972470284, |
| "num_tokens": 12796.0, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.06811409110259685, |
| "grad_norm": 0.7724414467811584, |
| "learning_rate": 0.00019544937428896475, |
| "loss": 1.3324, |
| "mean_token_accuracy": 0.8117582932114601, |
| "num_tokens": 17119.0, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.08514261387824607, |
| "grad_norm": 0.785762369632721, |
| "learning_rate": 0.00019431171786120593, |
| "loss": 1.3375, |
| "mean_token_accuracy": 0.8029867172241211, |
| "num_tokens": 21499.0, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.10217113665389528, |
| "grad_norm": 0.6702725291252136, |
| "learning_rate": 0.00019317406143344712, |
| "loss": 1.2159, |
| "mean_token_accuracy": 0.8230627462267875, |
| "num_tokens": 25785.0, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.11919965942954448, |
| "grad_norm": 0.6287369728088379, |
| "learning_rate": 0.0001920364050056883, |
| "loss": 1.1634, |
| "mean_token_accuracy": 0.8315445825457572, |
| "num_tokens": 29980.0, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.1362281822051937, |
| "grad_norm": 0.6573143005371094, |
| "learning_rate": 0.00019089874857792946, |
| "loss": 1.3762, |
| "mean_token_accuracy": 0.7971536681056023, |
| "num_tokens": 34497.0, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.1532567049808429, |
| "grad_norm": 0.6560688614845276, |
| "learning_rate": 0.00018976109215017067, |
| "loss": 1.1589, |
| "mean_token_accuracy": 0.8251897826790809, |
| "num_tokens": 38800.0, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.17028522775649213, |
| "grad_norm": 0.6654791831970215, |
| "learning_rate": 0.00018862343572241183, |
| "loss": 1.1498, |
| "mean_token_accuracy": 0.8319168403744698, |
| "num_tokens": 43055.0, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.18731375053214133, |
| "grad_norm": 0.7387417554855347, |
| "learning_rate": 0.00018748577929465302, |
| "loss": 1.1265, |
| "mean_token_accuracy": 0.8290244281291962, |
| "num_tokens": 47346.0, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.20434227330779056, |
| "grad_norm": 0.5795997977256775, |
| "learning_rate": 0.0001863481228668942, |
| "loss": 1.0046, |
| "mean_token_accuracy": 0.8534704640507698, |
| "num_tokens": 51485.0, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.22137079608343976, |
| "grad_norm": 0.6834130883216858, |
| "learning_rate": 0.0001852104664391354, |
| "loss": 1.113, |
| "mean_token_accuracy": 0.8345914751291275, |
| "num_tokens": 55716.0, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.23839931885908897, |
| "grad_norm": 0.7211284041404724, |
| "learning_rate": 0.00018407281001137657, |
| "loss": 1.2132, |
| "mean_token_accuracy": 0.8207321017980576, |
| "num_tokens": 60030.0, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.2554278416347382, |
| "grad_norm": 0.6312271356582642, |
| "learning_rate": 0.00018293515358361776, |
| "loss": 1.0987, |
| "mean_token_accuracy": 0.8333725541830063, |
| "num_tokens": 64288.0, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.2724563644103874, |
| "grad_norm": 0.7119404077529907, |
| "learning_rate": 0.00018179749715585894, |
| "loss": 1.0981, |
| "mean_token_accuracy": 0.8323698997497558, |
| "num_tokens": 68548.0, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.2894848871860366, |
| "grad_norm": 0.9000201225280762, |
| "learning_rate": 0.00018065984072810013, |
| "loss": 1.146, |
| "mean_token_accuracy": 0.8253335595130921, |
| "num_tokens": 72914.0, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.3065134099616858, |
| "grad_norm": 0.6050537824630737, |
| "learning_rate": 0.0001795221843003413, |
| "loss": 1.0111, |
| "mean_token_accuracy": 0.842674246430397, |
| "num_tokens": 77168.0, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.32354193273733506, |
| "grad_norm": 0.684512197971344, |
| "learning_rate": 0.0001783845278725825, |
| "loss": 1.1186, |
| "mean_token_accuracy": 0.8351956441998482, |
| "num_tokens": 81439.0, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.34057045551298426, |
| "grad_norm": 0.7323129773139954, |
| "learning_rate": 0.00017724687144482368, |
| "loss": 1.1287, |
| "mean_token_accuracy": 0.8245970487594605, |
| "num_tokens": 85773.0, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.35759897828863346, |
| "grad_norm": 0.7547741532325745, |
| "learning_rate": 0.00017610921501706487, |
| "loss": 1.0868, |
| "mean_token_accuracy": 0.8298466548323631, |
| "num_tokens": 90117.0, |
| "step": 105 |
| }, |
| { |
| "epoch": 0.37462750106428266, |
| "grad_norm": 0.7113769054412842, |
| "learning_rate": 0.00017497155858930602, |
| "loss": 1.1214, |
| "mean_token_accuracy": 0.8299670115113258, |
| "num_tokens": 94412.0, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.39165602383993187, |
| "grad_norm": 0.7486220002174377, |
| "learning_rate": 0.00017383390216154724, |
| "loss": 1.1452, |
| "mean_token_accuracy": 0.8283886179327965, |
| "num_tokens": 98782.0, |
| "step": 115 |
| }, |
| { |
| "epoch": 0.4086845466155811, |
| "grad_norm": 0.658633828163147, |
| "learning_rate": 0.0001726962457337884, |
| "loss": 1.0415, |
| "mean_token_accuracy": 0.8431179687380791, |
| "num_tokens": 103011.0, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.4257130693912303, |
| "grad_norm": 0.783794105052948, |
| "learning_rate": 0.0001715585893060296, |
| "loss": 1.1056, |
| "mean_token_accuracy": 0.8254878520965576, |
| "num_tokens": 107402.0, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.4427415921668795, |
| "grad_norm": 0.7954565286636353, |
| "learning_rate": 0.00017042093287827076, |
| "loss": 1.0518, |
| "mean_token_accuracy": 0.8375606715679169, |
| "num_tokens": 111707.0, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.45977011494252873, |
| "grad_norm": 0.6913949847221375, |
| "learning_rate": 0.00016928327645051198, |
| "loss": 1.1104, |
| "mean_token_accuracy": 0.8326424166560173, |
| "num_tokens": 115998.0, |
| "step": 135 |
| }, |
| { |
| "epoch": 0.47679863771817793, |
| "grad_norm": 0.7918037176132202, |
| "learning_rate": 0.00016814562002275313, |
| "loss": 1.0495, |
| "mean_token_accuracy": 0.8390962019562721, |
| "num_tokens": 120271.0, |
| "step": 140 |
| }, |
| { |
| "epoch": 0.49382716049382713, |
| "grad_norm": 0.7816826701164246, |
| "learning_rate": 0.00016700796359499432, |
| "loss": 1.0463, |
| "mean_token_accuracy": 0.8433194428682327, |
| "num_tokens": 124461.0, |
| "step": 145 |
| }, |
| { |
| "epoch": 0.5108556832694764, |
| "grad_norm": 0.8069042563438416, |
| "learning_rate": 0.0001658703071672355, |
| "loss": 1.0664, |
| "mean_token_accuracy": 0.8278923153877258, |
| "num_tokens": 128781.0, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.5278842060451255, |
| "grad_norm": 1.0114731788635254, |
| "learning_rate": 0.0001647326507394767, |
| "loss": 1.1217, |
| "mean_token_accuracy": 0.8267972275614739, |
| "num_tokens": 133167.0, |
| "step": 155 |
| }, |
| { |
| "epoch": 0.5449127288207748, |
| "grad_norm": 0.7900378704071045, |
| "learning_rate": 0.00016359499431171787, |
| "loss": 0.971, |
| "mean_token_accuracy": 0.8421565622091294, |
| "num_tokens": 137464.0, |
| "step": 160 |
| }, |
| { |
| "epoch": 0.561941251596424, |
| "grad_norm": 0.8059927821159363, |
| "learning_rate": 0.00016245733788395906, |
| "loss": 1.0457, |
| "mean_token_accuracy": 0.8381335601210594, |
| "num_tokens": 141760.0, |
| "step": 165 |
| }, |
| { |
| "epoch": 0.5789697743720732, |
| "grad_norm": 0.7589371800422668, |
| "learning_rate": 0.00016131968145620024, |
| "loss": 1.0208, |
| "mean_token_accuracy": 0.8342296928167343, |
| "num_tokens": 146051.0, |
| "step": 170 |
| }, |
| { |
| "epoch": 0.5959982971477225, |
| "grad_norm": 0.692416250705719, |
| "learning_rate": 0.00016018202502844143, |
| "loss": 0.9837, |
| "mean_token_accuracy": 0.8510025009512902, |
| "num_tokens": 150270.0, |
| "step": 175 |
| }, |
| { |
| "epoch": 0.6130268199233716, |
| "grad_norm": 0.6662923097610474, |
| "learning_rate": 0.0001590443686006826, |
| "loss": 0.9835, |
| "mean_token_accuracy": 0.8463438332080842, |
| "num_tokens": 154537.0, |
| "step": 180 |
| }, |
| { |
| "epoch": 0.6300553426990209, |
| "grad_norm": 0.7452691793441772, |
| "learning_rate": 0.0001579067121729238, |
| "loss": 1.0496, |
| "mean_token_accuracy": 0.8375737622380257, |
| "num_tokens": 158810.0, |
| "step": 185 |
| }, |
| { |
| "epoch": 0.6470838654746701, |
| "grad_norm": 0.809700608253479, |
| "learning_rate": 0.00015676905574516496, |
| "loss": 1.0595, |
| "mean_token_accuracy": 0.8294038504362107, |
| "num_tokens": 163213.0, |
| "step": 190 |
| }, |
| { |
| "epoch": 0.6641123882503193, |
| "grad_norm": 0.7708745002746582, |
| "learning_rate": 0.00015563139931740617, |
| "loss": 1.0976, |
| "mean_token_accuracy": 0.8289313957095146, |
| "num_tokens": 167607.0, |
| "step": 195 |
| }, |
| { |
| "epoch": 0.6811409110259685, |
| "grad_norm": 0.6482141613960266, |
| "learning_rate": 0.00015449374288964733, |
| "loss": 1.0701, |
| "mean_token_accuracy": 0.8261386752128601, |
| "num_tokens": 172024.0, |
| "step": 200 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 879, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 3, |
| "save_steps": 200, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 1612393756090368.0, |
| "train_batch_size": 1, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|