Instructions to use FormlessAI/Qwen2.5-3B-Instruct-Reasoning-TogetherAI with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use FormlessAI/Qwen2.5-3B-Instruct-Reasoning-TogetherAI with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen2.5-3B-Instruct") model = PeftModel.from_pretrained(base_model, "FormlessAI/Qwen2.5-3B-Instruct-Reasoning-TogetherAI") - Notebooks
- Google Colab
- Kaggle
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 2.0, | |
| "eval_steps": 56, | |
| "global_step": 56, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.03571428571428571, | |
| "grad_norm": 2.694387197494507, | |
| "learning_rate": 1e-05, | |
| "loss": 2.857421875, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.07142857142857142, | |
| "grad_norm": 2.558368682861328, | |
| "learning_rate": 9.992134075089085e-06, | |
| "loss": 2.81640625, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.10714285714285714, | |
| "grad_norm": 2.3488693237304688, | |
| "learning_rate": 9.968561049466214e-06, | |
| "loss": 2.681640625, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.14285714285714285, | |
| "grad_norm": 2.1087405681610107, | |
| "learning_rate": 9.92935509259118e-06, | |
| "loss": 2.544921875, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.17857142857142858, | |
| "grad_norm": 1.6803760528564453, | |
| "learning_rate": 9.874639560909118e-06, | |
| "loss": 2.330078125, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.21428571428571427, | |
| "grad_norm": 1.655576229095459, | |
| "learning_rate": 9.804586609725499e-06, | |
| "loss": 2.263671875, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.25, | |
| "grad_norm": 1.5180833339691162, | |
| "learning_rate": 9.719416651541839e-06, | |
| "loss": 2.1240234375, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.2857142857142857, | |
| "grad_norm": 1.6885188817977905, | |
| "learning_rate": 9.619397662556434e-06, | |
| "loss": 2.09814453125, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.32142857142857145, | |
| "grad_norm": 1.3130377531051636, | |
| "learning_rate": 9.504844339512096e-06, | |
| "loss": 1.8671875, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.35714285714285715, | |
| "grad_norm": 0.7923501133918762, | |
| "learning_rate": 9.376117109543769e-06, | |
| "loss": 1.8203125, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.39285714285714285, | |
| "grad_norm": 0.6045828461647034, | |
| "learning_rate": 9.233620996141421e-06, | |
| "loss": 1.775390625, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.42857142857142855, | |
| "grad_norm": 0.509148895740509, | |
| "learning_rate": 9.077804344796302e-06, | |
| "loss": 1.7275390625, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.4642857142857143, | |
| "grad_norm": 0.5257807970046997, | |
| "learning_rate": 8.90915741234015e-06, | |
| "loss": 1.7373046875, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.5, | |
| "grad_norm": 0.5001201629638672, | |
| "learning_rate": 8.728210824415829e-06, | |
| "loss": 1.708984375, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.5357142857142857, | |
| "grad_norm": 0.4635830819606781, | |
| "learning_rate": 8.535533905932739e-06, | |
| "loss": 1.6552734375, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.5714285714285714, | |
| "grad_norm": 0.36223411560058594, | |
| "learning_rate": 8.331732889760021e-06, | |
| "loss": 1.666015625, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.6071428571428571, | |
| "grad_norm": 0.392349511384964, | |
| "learning_rate": 8.117449009293668e-06, | |
| "loss": 1.61328125, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.6428571428571429, | |
| "grad_norm": 0.3374916911125183, | |
| "learning_rate": 7.89335648089903e-06, | |
| "loss": 1.595703125, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.6785714285714286, | |
| "grad_norm": 0.2974811792373657, | |
| "learning_rate": 7.660160382576683e-06, | |
| "loss": 1.57421875, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.7142857142857143, | |
| "grad_norm": 0.30373185873031616, | |
| "learning_rate": 7.4185944355261996e-06, | |
| "loss": 1.546875, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.75, | |
| "grad_norm": 0.3105020821094513, | |
| "learning_rate": 7.169418695587791e-06, | |
| "loss": 1.513671875, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.7857142857142857, | |
| "grad_norm": 0.33336368203163147, | |
| "learning_rate": 6.913417161825449e-06, | |
| "loss": 1.5361328125, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 0.8214285714285714, | |
| "grad_norm": 0.3261295557022095, | |
| "learning_rate": 6.651395309775837e-06, | |
| "loss": 1.4931640625, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 0.8571428571428571, | |
| "grad_norm": 0.3157199025154114, | |
| "learning_rate": 6.384177557124247e-06, | |
| "loss": 1.48779296875, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.8928571428571429, | |
| "grad_norm": 0.3149335980415344, | |
| "learning_rate": 6.112604669781572e-06, | |
| "loss": 1.46875, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.9285714285714286, | |
| "grad_norm": 0.3011397421360016, | |
| "learning_rate": 5.837531116523683e-06, | |
| "loss": 1.4775390625, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 0.9642857142857143, | |
| "grad_norm": 0.3071064054965973, | |
| "learning_rate": 5.559822380516539e-06, | |
| "loss": 1.44921875, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 1.0, | |
| "grad_norm": 0.29467642307281494, | |
| "learning_rate": 5.2803522361859596e-06, | |
| "loss": 1.44091796875, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 1.0357142857142858, | |
| "grad_norm": 0.3383062481880188, | |
| "learning_rate": 5e-06, | |
| "loss": 1.4287109375, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 1.0714285714285714, | |
| "grad_norm": 0.3466164469718933, | |
| "learning_rate": 4.719647763814041e-06, | |
| "loss": 1.4345703125, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 1.1071428571428572, | |
| "grad_norm": 0.33726248145103455, | |
| "learning_rate": 4.4401776194834615e-06, | |
| "loss": 1.4052734375, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 1.1428571428571428, | |
| "grad_norm": 0.36348649859428406, | |
| "learning_rate": 4.162468883476319e-06, | |
| "loss": 1.35693359375, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 1.1785714285714286, | |
| "grad_norm": 0.31096431612968445, | |
| "learning_rate": 3.887395330218429e-06, | |
| "loss": 1.38671875, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 1.2142857142857142, | |
| "grad_norm": 0.3586256802082062, | |
| "learning_rate": 3.6158224428757538e-06, | |
| "loss": 1.3544921875, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 1.25, | |
| "grad_norm": 0.3652304708957672, | |
| "learning_rate": 3.3486046902241663e-06, | |
| "loss": 1.37890625, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 1.2857142857142856, | |
| "grad_norm": 0.4301171600818634, | |
| "learning_rate": 3.0865828381745515e-06, | |
| "loss": 1.39013671875, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 1.3214285714285714, | |
| "grad_norm": 0.38166680932044983, | |
| "learning_rate": 2.83058130441221e-06, | |
| "loss": 1.357421875, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 1.3571428571428572, | |
| "grad_norm": 0.3681507706642151, | |
| "learning_rate": 2.5814055644738013e-06, | |
| "loss": 1.3408203125, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 1.3928571428571428, | |
| "grad_norm": 0.34061020612716675, | |
| "learning_rate": 2.339839617423318e-06, | |
| "loss": 1.341796875, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 1.4285714285714286, | |
| "grad_norm": 0.2774454951286316, | |
| "learning_rate": 2.1066435191009717e-06, | |
| "loss": 1.3369140625, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 1.4642857142857144, | |
| "grad_norm": 0.2526361346244812, | |
| "learning_rate": 1.8825509907063328e-06, | |
| "loss": 1.34423828125, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 1.5, | |
| "grad_norm": 0.24453622102737427, | |
| "learning_rate": 1.6682671102399806e-06, | |
| "loss": 1.3115234375, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 1.5357142857142856, | |
| "grad_norm": 0.2164342850446701, | |
| "learning_rate": 1.4644660940672628e-06, | |
| "loss": 1.28076171875, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 1.5714285714285714, | |
| "grad_norm": 0.19291897118091583, | |
| "learning_rate": 1.2717891755841722e-06, | |
| "loss": 1.359375, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 1.6071428571428572, | |
| "grad_norm": 0.21438544988632202, | |
| "learning_rate": 1.0908425876598512e-06, | |
| "loss": 1.2685546875, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 1.6428571428571428, | |
| "grad_norm": 0.1777007132768631, | |
| "learning_rate": 9.221956552036992e-07, | |
| "loss": 1.296875, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 1.6785714285714286, | |
| "grad_norm": 0.16975992918014526, | |
| "learning_rate": 7.663790038585794e-07, | |
| "loss": 1.3310546875, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 1.7142857142857144, | |
| "grad_norm": 0.15860167145729065, | |
| "learning_rate": 6.238828904562316e-07, | |
| "loss": 1.30859375, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 1.75, | |
| "grad_norm": 0.1576526165008545, | |
| "learning_rate": 4.951556604879049e-07, | |
| "loss": 1.29052734375, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 1.7857142857142856, | |
| "grad_norm": 0.16139984130859375, | |
| "learning_rate": 3.8060233744356634e-07, | |
| "loss": 1.3203125, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 1.8214285714285714, | |
| "grad_norm": 0.15369921922683716, | |
| "learning_rate": 2.8058334845816214e-07, | |
| "loss": 1.2919921875, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 1.8571428571428572, | |
| "grad_norm": 0.1540217250585556, | |
| "learning_rate": 1.9541339027450256e-07, | |
| "loss": 1.3134765625, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 1.8928571428571428, | |
| "grad_norm": 0.14983929693698883, | |
| "learning_rate": 1.253604390908819e-07, | |
| "loss": 1.310546875, | |
| "step": 53 | |
| }, | |
| { | |
| "epoch": 1.9285714285714286, | |
| "grad_norm": 0.14664843678474426, | |
| "learning_rate": 7.064490740882057e-08, | |
| "loss": 1.33740234375, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 1.9642857142857144, | |
| "grad_norm": 0.1502656489610672, | |
| "learning_rate": 3.143895053378698e-08, | |
| "loss": 1.3173828125, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "grad_norm": 0.154768168926239, | |
| "learning_rate": 7.865924910916977e-09, | |
| "loss": 1.3291015625, | |
| "step": 56 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "eval_loss": 1.3828125, | |
| "eval_runtime": 1.0297, | |
| "eval_samples_per_second": 9.712, | |
| "eval_steps_per_second": 0.971, | |
| "step": 56 | |
| } | |
| ], | |
| "logging_steps": 1.0, | |
| "max_steps": 56, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 2, | |
| "save_steps": 0, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1.0007198831025848e+18, | |
| "train_batch_size": 4, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |