Qwen2.5-Coder-Text2Sql-ft / workspace /fine_tunning /text2sql /output_train2 /checkpoint-2000 /trainer_state.json
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 4.444444444444445, | |
| "eval_steps": 500, | |
| "global_step": 2000, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.1111111111111111, | |
| "grad_norm": 0.20243267714977264, | |
| "learning_rate": 5e-05, | |
| "loss": 0.1508, | |
| "mean_token_accuracy": 0.9614167541265488, | |
| "num_tokens": 172590.0, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.2222222222222222, | |
| "grad_norm": 0.22473949193954468, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0762, | |
| "mean_token_accuracy": 0.974150316119194, | |
| "num_tokens": 346277.0, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.3333333333333333, | |
| "grad_norm": 0.1306583136320114, | |
| "learning_rate": 5e-05, | |
| "loss": 0.059, | |
| "mean_token_accuracy": 0.9791698586940766, | |
| "num_tokens": 521548.0, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.4444444444444444, | |
| "grad_norm": 0.2091003954410553, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0511, | |
| "mean_token_accuracy": 0.9820115834474563, | |
| "num_tokens": 694583.0, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.5555555555555556, | |
| "grad_norm": 0.18803296983242035, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0443, | |
| "mean_token_accuracy": 0.9839017766714097, | |
| "num_tokens": 868203.0, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.6666666666666666, | |
| "grad_norm": 0.19305875897407532, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0484, | |
| "mean_token_accuracy": 0.9826253122091293, | |
| "num_tokens": 1044270.0, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.7777777777777778, | |
| "grad_norm": 0.16429616510868073, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0452, | |
| "mean_token_accuracy": 0.9832556092739105, | |
| "num_tokens": 1221728.0, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.8888888888888888, | |
| "grad_norm": 0.17009682953357697, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0419, | |
| "mean_token_accuracy": 0.9852031743526459, | |
| "num_tokens": 1395873.0, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 1.0, | |
| "grad_norm": 0.14838816225528717, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0458, | |
| "mean_token_accuracy": 0.9842678928375244, | |
| "num_tokens": 1570780.0, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 1.1111111111111112, | |
| "grad_norm": 0.20167161524295807, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0335, | |
| "mean_token_accuracy": 0.9871636009216309, | |
| "num_tokens": 1746181.0, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 1.2222222222222223, | |
| "grad_norm": 0.22028109431266785, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0338, | |
| "mean_token_accuracy": 0.9862233906984329, | |
| "num_tokens": 1921414.0, | |
| "step": 550 | |
| }, | |
| { | |
| "epoch": 1.3333333333333333, | |
| "grad_norm": 0.12278841435909271, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0299, | |
| "mean_token_accuracy": 0.9886959612369537, | |
| "num_tokens": 2095976.0, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 1.4444444444444444, | |
| "grad_norm": 0.09097229689359665, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0344, | |
| "mean_token_accuracy": 0.9869739925861358, | |
| "num_tokens": 2269976.0, | |
| "step": 650 | |
| }, | |
| { | |
| "epoch": 1.5555555555555556, | |
| "grad_norm": 0.149724543094635, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0312, | |
| "mean_token_accuracy": 0.9881853342056275, | |
| "num_tokens": 2444238.0, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 1.6666666666666665, | |
| "grad_norm": 0.1601683348417282, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0375, | |
| "mean_token_accuracy": 0.9859996843338013, | |
| "num_tokens": 2618992.0, | |
| "step": 750 | |
| }, | |
| { | |
| "epoch": 1.7777777777777777, | |
| "grad_norm": 0.14399662613868713, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0273, | |
| "mean_token_accuracy": 0.9900204366445542, | |
| "num_tokens": 2793057.0, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 1.8888888888888888, | |
| "grad_norm": 0.1325361728668213, | |
| "learning_rate": 5e-05, | |
| "loss": 0.037, | |
| "mean_token_accuracy": 0.9859159809350967, | |
| "num_tokens": 2966769.0, | |
| "step": 850 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "grad_norm": 0.12275587767362595, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0359, | |
| "mean_token_accuracy": 0.9860490292310715, | |
| "num_tokens": 3141560.0, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 2.111111111111111, | |
| "grad_norm": 0.14534829556941986, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0238, | |
| "mean_token_accuracy": 0.9910678189992904, | |
| "num_tokens": 3315113.0, | |
| "step": 950 | |
| }, | |
| { | |
| "epoch": 2.2222222222222223, | |
| "grad_norm": 0.1087908148765564, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0281, | |
| "mean_token_accuracy": 0.9884923130273819, | |
| "num_tokens": 3488791.0, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 2.3333333333333335, | |
| "grad_norm": 0.16228517889976501, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0234, | |
| "mean_token_accuracy": 0.9911258590221405, | |
| "num_tokens": 3663372.0, | |
| "step": 1050 | |
| }, | |
| { | |
| "epoch": 2.4444444444444446, | |
| "grad_norm": 0.05340925231575966, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0254, | |
| "mean_token_accuracy": 0.9899671649932862, | |
| "num_tokens": 3837397.0, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 2.5555555555555554, | |
| "grad_norm": 0.12148639559745789, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0296, | |
| "mean_token_accuracy": 0.9886352431774139, | |
| "num_tokens": 4012362.0, | |
| "step": 1150 | |
| }, | |
| { | |
| "epoch": 2.6666666666666665, | |
| "grad_norm": 0.09275887161493301, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0269, | |
| "mean_token_accuracy": 0.989583585858345, | |
| "num_tokens": 4188410.0, | |
| "step": 1200 | |
| }, | |
| { | |
| "epoch": 2.7777777777777777, | |
| "grad_norm": 0.17173512279987335, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0275, | |
| "mean_token_accuracy": 0.9893661296367645, | |
| "num_tokens": 4364549.0, | |
| "step": 1250 | |
| }, | |
| { | |
| "epoch": 2.888888888888889, | |
| "grad_norm": 0.18386530876159668, | |
| "learning_rate": 5e-05, | |
| "loss": 0.03, | |
| "mean_token_accuracy": 0.9877435737848281, | |
| "num_tokens": 4538708.0, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 3.0, | |
| "grad_norm": 0.0737985298037529, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0256, | |
| "mean_token_accuracy": 0.9906050026416778, | |
| "num_tokens": 4712340.0, | |
| "step": 1350 | |
| }, | |
| { | |
| "epoch": 3.111111111111111, | |
| "grad_norm": 0.10669398307800293, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0231, | |
| "mean_token_accuracy": 0.9914622634649277, | |
| "num_tokens": 4887386.0, | |
| "step": 1400 | |
| }, | |
| { | |
| "epoch": 3.2222222222222223, | |
| "grad_norm": 0.09974446147680283, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0212, | |
| "mean_token_accuracy": 0.9920986944437027, | |
| "num_tokens": 5059979.0, | |
| "step": 1450 | |
| }, | |
| { | |
| "epoch": 3.3333333333333335, | |
| "grad_norm": 0.18398422002792358, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0229, | |
| "mean_token_accuracy": 0.9908855521678924, | |
| "num_tokens": 5233989.0, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 3.4444444444444446, | |
| "grad_norm": 0.10202126950025558, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0236, | |
| "mean_token_accuracy": 0.9902925211191177, | |
| "num_tokens": 5410888.0, | |
| "step": 1550 | |
| }, | |
| { | |
| "epoch": 3.5555555555555554, | |
| "grad_norm": 0.1483769565820694, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0244, | |
| "mean_token_accuracy": 0.9899595558643342, | |
| "num_tokens": 5584175.0, | |
| "step": 1600 | |
| }, | |
| { | |
| "epoch": 3.6666666666666665, | |
| "grad_norm": 0.07508880645036697, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0226, | |
| "mean_token_accuracy": 0.9916278702020646, | |
| "num_tokens": 5759747.0, | |
| "step": 1650 | |
| }, | |
| { | |
| "epoch": 3.7777777777777777, | |
| "grad_norm": 0.08631623536348343, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0252, | |
| "mean_token_accuracy": 0.9896138072013855, | |
| "num_tokens": 5933352.0, | |
| "step": 1700 | |
| }, | |
| { | |
| "epoch": 3.888888888888889, | |
| "grad_norm": 0.12719225883483887, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0213, | |
| "mean_token_accuracy": 0.9918474739789963, | |
| "num_tokens": 6108120.0, | |
| "step": 1750 | |
| }, | |
| { | |
| "epoch": 4.0, | |
| "grad_norm": 0.07570526748895645, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0243, | |
| "mean_token_accuracy": 0.990051948428154, | |
| "num_tokens": 6283120.0, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 4.111111111111111, | |
| "grad_norm": 0.7129213213920593, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0181, | |
| "mean_token_accuracy": 0.9931919860839844, | |
| "num_tokens": 6456943.0, | |
| "step": 1850 | |
| }, | |
| { | |
| "epoch": 4.222222222222222, | |
| "grad_norm": 0.132768452167511, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0202, | |
| "mean_token_accuracy": 0.9916618072986603, | |
| "num_tokens": 6632907.0, | |
| "step": 1900 | |
| }, | |
| { | |
| "epoch": 4.333333333333333, | |
| "grad_norm": 0.1010010689496994, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0212, | |
| "mean_token_accuracy": 0.9916968619823456, | |
| "num_tokens": 6805656.0, | |
| "step": 1950 | |
| }, | |
| { | |
| "epoch": 4.444444444444445, | |
| "grad_norm": 0.22385317087173462, | |
| "learning_rate": 5e-05, | |
| "loss": 0.0204, | |
| "mean_token_accuracy": 0.9914450365304946, | |
| "num_tokens": 6978299.0, | |
| "step": 2000 | |
| } | |
| ], | |
| "logging_steps": 50, | |
| "max_steps": 2250, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 5, | |
| "save_steps": 500, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 3.174835230239662e+17, | |
| "train_batch_size": 2, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |