Instructions to use FormlessAI/Qwen3.5-0.8B-Reasoning-TogetherAI with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use FormlessAI/Qwen3.5-0.8B-Reasoning-TogetherAI with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("togethercomputer/Qwen3.5-0.8B") model = PeftModel.from_pretrained(base_model, "FormlessAI/Qwen3.5-0.8B-Reasoning-TogetherAI") - Notebooks
- Google Colab
- Kaggle
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 2.0, | |
| "eval_steps": 52, | |
| "global_step": 52, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.038461538461538464, | |
| "grad_norm": 13.78187084197998, | |
| "learning_rate": 1e-05, | |
| "loss": 2.6171875, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.07692307692307693, | |
| "grad_norm": 9.845561981201172, | |
| "learning_rate": 9.990877771116588e-06, | |
| "loss": 2.373046875, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.11538461538461539, | |
| "grad_norm": 8.01469898223877, | |
| "learning_rate": 9.96354437049027e-06, | |
| "loss": 2.2021484375, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.15384615384615385, | |
| "grad_norm": 5.164867877960205, | |
| "learning_rate": 9.91809953473572e-06, | |
| "loss": 1.9962158203125, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.19230769230769232, | |
| "grad_norm": 4.263365268707275, | |
| "learning_rate": 9.854709087130261e-06, | |
| "loss": 1.894287109375, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.23076923076923078, | |
| "grad_norm": 3.3784377574920654, | |
| "learning_rate": 9.77360433254273e-06, | |
| "loss": 1.86181640625, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.2692307692307692, | |
| "grad_norm": 2.9527230262756348, | |
| "learning_rate": 9.675081213427076e-06, | |
| "loss": 1.834716796875, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.3076923076923077, | |
| "grad_norm": 2.2422053813934326, | |
| "learning_rate": 9.55949922996045e-06, | |
| "loss": 1.65478515625, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.34615384615384615, | |
| "grad_norm": 1.7632695436477661, | |
| "learning_rate": 9.427280128266049e-06, | |
| "loss": 1.589111328125, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.38461538461538464, | |
| "grad_norm": 1.6509746313095093, | |
| "learning_rate": 9.278906361507238e-06, | |
| "loss": 1.593994140625, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.4230769230769231, | |
| "grad_norm": 1.3932156562805176, | |
| "learning_rate": 9.114919329468283e-06, | |
| "loss": 1.4892578125, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.46153846153846156, | |
| "grad_norm": 1.2194513082504272, | |
| "learning_rate": 8.935917403045251e-06, | |
| "loss": 1.54052734375, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.5, | |
| "grad_norm": 0.9014191031455994, | |
| "learning_rate": 8.742553740855507e-06, | |
| "loss": 1.5517578125, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.5384615384615384, | |
| "grad_norm": 0.7796593904495239, | |
| "learning_rate": 8.535533905932739e-06, | |
| "loss": 1.50323486328125, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.5769230769230769, | |
| "grad_norm": 0.6279335021972656, | |
| "learning_rate": 8.315613291203977e-06, | |
| "loss": 1.46630859375, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.6153846153846154, | |
| "grad_norm": 0.6258265972137451, | |
| "learning_rate": 8.083594363142717e-06, | |
| "loss": 1.45782470703125, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.6538461538461539, | |
| "grad_norm": 0.5115531086921692, | |
| "learning_rate": 7.84032373365578e-06, | |
| "loss": 1.43798828125, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.6923076923076923, | |
| "grad_norm": 0.4573651850223541, | |
| "learning_rate": 7.586689070888284e-06, | |
| "loss": 1.4058837890625, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.7307692307692307, | |
| "grad_norm": 0.46966230869293213, | |
| "learning_rate": 7.323615860218844e-06, | |
| "loss": 1.47265625, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.7692307692307693, | |
| "grad_norm": 0.3782135546207428, | |
| "learning_rate": 7.052064027263785e-06, | |
| "loss": 1.4276123046875, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.8076923076923077, | |
| "grad_norm": 0.3979575037956238, | |
| "learning_rate": 6.773024435212678e-06, | |
| "loss": 1.484375, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.8461538461538461, | |
| "grad_norm": 0.41323113441467285, | |
| "learning_rate": 6.487515269276015e-06, | |
| "loss": 1.370269775390625, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 0.8846153846153846, | |
| "grad_norm": 0.384005606174469, | |
| "learning_rate": 6.1965783214377895e-06, | |
| "loss": 1.41943359375, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 0.9230769230769231, | |
| "grad_norm": 0.3821006715297699, | |
| "learning_rate": 5.90127518906953e-06, | |
| "loss": 1.39990234375, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.9615384615384616, | |
| "grad_norm": 0.385856032371521, | |
| "learning_rate": 5.6026834012766155e-06, | |
| "loss": 1.47509765625, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 1.0, | |
| "grad_norm": 0.32634466886520386, | |
| "learning_rate": 5.301892487111431e-06, | |
| "loss": 1.41748046875, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 1.0384615384615385, | |
| "grad_norm": 0.3082049489021301, | |
| "learning_rate": 5e-06, | |
| "loss": 1.37939453125, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 1.0769230769230769, | |
| "grad_norm": 0.2960004210472107, | |
| "learning_rate": 4.69810751288857e-06, | |
| "loss": 1.431640625, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 1.1153846153846154, | |
| "grad_norm": 0.2712954878807068, | |
| "learning_rate": 4.397316598723385e-06, | |
| "loss": 1.39599609375, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 1.1538461538461537, | |
| "grad_norm": 0.2660493552684784, | |
| "learning_rate": 4.098724810930472e-06, | |
| "loss": 1.3868408203125, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 1.1923076923076923, | |
| "grad_norm": 0.2757558822631836, | |
| "learning_rate": 3.803421678562213e-06, | |
| "loss": 1.3758544921875, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 1.2307692307692308, | |
| "grad_norm": 0.26696985960006714, | |
| "learning_rate": 3.5124847307239863e-06, | |
| "loss": 1.4422607421875, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 1.2692307692307692, | |
| "grad_norm": 0.29158878326416016, | |
| "learning_rate": 3.226975564787322e-06, | |
| "loss": 1.507568359375, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 1.3076923076923077, | |
| "grad_norm": 0.2519815266132355, | |
| "learning_rate": 2.947935972736217e-06, | |
| "loss": 1.3935546875, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 1.3461538461538463, | |
| "grad_norm": 0.24109189212322235, | |
| "learning_rate": 2.6763841397811576e-06, | |
| "loss": 1.364501953125, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 1.3846153846153846, | |
| "grad_norm": 0.2500540316104889, | |
| "learning_rate": 2.4133109291117156e-06, | |
| "loss": 1.392822265625, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 1.4230769230769231, | |
| "grad_norm": 0.35815808176994324, | |
| "learning_rate": 2.159676266344222e-06, | |
| "loss": 1.31640625, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 1.4615384615384617, | |
| "grad_norm": 0.299280047416687, | |
| "learning_rate": 1.9164056368572847e-06, | |
| "loss": 1.385498046875, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 1.5, | |
| "grad_norm": 0.263439804315567, | |
| "learning_rate": 1.6843867087960252e-06, | |
| "loss": 1.42724609375, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 1.5384615384615383, | |
| "grad_norm": 0.22188825905323029, | |
| "learning_rate": 1.4644660940672628e-06, | |
| "loss": 1.391357421875, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 1.5769230769230769, | |
| "grad_norm": 0.23454146087169647, | |
| "learning_rate": 1.257446259144494e-06, | |
| "loss": 1.3671875, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 1.6153846153846154, | |
| "grad_norm": 0.2985825538635254, | |
| "learning_rate": 1.0640825969547498e-06, | |
| "loss": 1.36846923828125, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 1.6538461538461537, | |
| "grad_norm": 0.21381421387195587, | |
| "learning_rate": 8.850806705317183e-07, | |
| "loss": 1.3594970703125, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 1.6923076923076923, | |
| "grad_norm": 0.21711355447769165, | |
| "learning_rate": 7.210936384927631e-07, | |
| "loss": 1.3319091796875, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 1.7307692307692308, | |
| "grad_norm": 0.2467353194952011, | |
| "learning_rate": 5.727198717339511e-07, | |
| "loss": 1.40771484375, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 1.7692307692307692, | |
| "grad_norm": 0.22233083844184875, | |
| "learning_rate": 4.405007700395497e-07, | |
| "loss": 1.37060546875, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 1.8076923076923077, | |
| "grad_norm": 0.2567121088504791, | |
| "learning_rate": 3.2491878657292643e-07, | |
| "loss": 1.43212890625, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 1.8461538461538463, | |
| "grad_norm": 0.25618961453437805, | |
| "learning_rate": 2.2639566745727203e-07, | |
| "loss": 1.32318115234375, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 1.8846153846153846, | |
| "grad_norm": 0.2310512363910675, | |
| "learning_rate": 1.4529091286973994e-07, | |
| "loss": 1.375732421875, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 1.9230769230769231, | |
| "grad_norm": 0.2249426245689392, | |
| "learning_rate": 8.190046526428241e-08, | |
| "loss": 1.361328125, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 1.9615384615384617, | |
| "grad_norm": 0.25166356563568115, | |
| "learning_rate": 3.645562950973014e-08, | |
| "loss": 1.4404296875, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "grad_norm": 0.21272313594818115, | |
| "learning_rate": 9.12222888341252e-09, | |
| "loss": 1.3857421875, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "eval_loss": 1.439453125, | |
| "eval_runtime": 9.74, | |
| "eval_samples_per_second": 0.308, | |
| "eval_steps_per_second": 0.103, | |
| "step": 52 | |
| } | |
| ], | |
| "logging_steps": 1.0, | |
| "max_steps": 52, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 2, | |
| "save_steps": 0, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 2.0288909949167206e+17, | |
| "train_batch_size": 1, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |