Instructions to use VQA-DeepLearning/gemma_2_lora_E2b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use VQA-DeepLearning/gemma_2_lora_E2b with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("VQA-DeepLearning/gemma_2_lora_E2b", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Unsloth Studio
How to use VQA-DeepLearning/gemma_2_lora_E2b with Unsloth Studio:
Install Unsloth Studio (macOS, Linux, WSL)
curl -fsSL https://unsloth.ai/install.sh | sh # Run unsloth studio unsloth studio -H 0.0.0.0 -p 8888 # Then open http://localhost:8888 in your browser # Search for VQA-DeepLearning/gemma_2_lora_E2b to start chatting
Install Unsloth Studio (Windows)
irm https://unsloth.ai/install.ps1 | iex # Run unsloth studio unsloth studio -H 0.0.0.0 -p 8888 # Then open http://localhost:8888 in your browser # Search for VQA-DeepLearning/gemma_2_lora_E2b to start chatting
Using HuggingFace Spaces for Unsloth
# No setup required # Open https://huggingface.co/spaces/unsloth/studio in your browser # Search for VQA-DeepLearning/gemma_2_lora_E2b to start chatting
Load model with FastModel
pip install unsloth from unsloth import FastModel model, tokenizer = FastModel.from_pretrained( model_name="VQA-DeepLearning/gemma_2_lora_E2b", max_seq_length=2048, )
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 1.894590072504183, | |
| "eval_steps": 100, | |
| "global_step": 850, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.022308979364194088, | |
| "grad_norm": 3.6059792041778564, | |
| "learning_rate": 0.00019999016517595753, | |
| "loss": 3.6749305725097656, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.044617958728388175, | |
| "grad_norm": 1.686761498451233, | |
| "learning_rate": 0.00019987954562051725, | |
| "loss": 0.9972598075866699, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.06692693809258227, | |
| "grad_norm": 1.0181570053100586, | |
| "learning_rate": 0.00019964614941176195, | |
| "loss": 0.7280679225921631, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.08923591745677635, | |
| "grad_norm": 0.9052721261978149, | |
| "learning_rate": 0.00019929026345133122, | |
| "loss": 0.6308993339538574, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.11154489682097044, | |
| "grad_norm": 0.8318365812301636, | |
| "learning_rate": 0.00019881232521105089, | |
| "loss": 0.6105688571929931, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.13385387618516453, | |
| "grad_norm": 0.9255080223083496, | |
| "learning_rate": 0.00019821292219517192, | |
| "loss": 0.5463937282562256, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.1561628555493586, | |
| "grad_norm": 0.8712772727012634, | |
| "learning_rate": 0.00019749279121818235, | |
| "loss": 0.513142728805542, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.1784718349135527, | |
| "grad_norm": 0.6806803941726685, | |
| "learning_rate": 0.00019665281749908033, | |
| "loss": 0.45452089309692384, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.2007808142777468, | |
| "grad_norm": 0.8569662570953369, | |
| "learning_rate": 0.0001956940335732209, | |
| "loss": 0.4099240303039551, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.22308979364194087, | |
| "grad_norm": 0.5690306425094604, | |
| "learning_rate": 0.00019461761802307495, | |
| "loss": 0.4181380748748779, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.22308979364194087, | |
| "eval_loss": 7.142496585845947, | |
| "eval_runtime": 196.0856, | |
| "eval_samples_per_second": 2.3, | |
| "eval_steps_per_second": 2.3, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.24539877300613497, | |
| "grad_norm": 0.5885435938835144, | |
| "learning_rate": 0.00019342489402945998, | |
| "loss": 0.3779646873474121, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.26770775237032907, | |
| "grad_norm": 0.7473471164703369, | |
| "learning_rate": 0.00019211732774502372, | |
| "loss": 0.3981457710266113, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.29001673173452314, | |
| "grad_norm": 0.5102096796035767, | |
| "learning_rate": 0.00019069652649198005, | |
| "loss": 0.355985426902771, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.3123257110987172, | |
| "grad_norm": 0.9202075004577637, | |
| "learning_rate": 0.00018916423678631272, | |
| "loss": 0.4652230262756348, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.33463469046291133, | |
| "grad_norm": 0.47609537839889526, | |
| "learning_rate": 0.00018752234219087538, | |
| "loss": 0.31681218147277834, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.3569436698271054, | |
| "grad_norm": 0.6477280855178833, | |
| "learning_rate": 0.00018577286100002723, | |
| "loss": 0.414081335067749, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.3792526491912995, | |
| "grad_norm": 0.5019633769989014, | |
| "learning_rate": 0.00018391794375865024, | |
| "loss": 0.4289548873901367, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.4015616285554936, | |
| "grad_norm": 0.906410813331604, | |
| "learning_rate": 0.0001819598706185979, | |
| "loss": 0.33823583126068113, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.42387060791968767, | |
| "grad_norm": 0.5761931538581848, | |
| "learning_rate": 0.00017990104853582493, | |
| "loss": 0.3813004493713379, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.44617958728388174, | |
| "grad_norm": 0.49768441915512085, | |
| "learning_rate": 0.00017774400831164323, | |
| "loss": 0.39550683498382566, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.44617958728388174, | |
| "eval_loss": 6.169673442840576, | |
| "eval_runtime": 191.4793, | |
| "eval_samples_per_second": 2.355, | |
| "eval_steps_per_second": 2.355, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.46848856664807587, | |
| "grad_norm": 0.4040807783603668, | |
| "learning_rate": 0.0001754914014817416, | |
| "loss": 0.3376659870147705, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.49079754601226994, | |
| "grad_norm": 0.7800427675247192, | |
| "learning_rate": 0.00017314599705679277, | |
| "loss": 0.34065258502960205, | |
| "step": 220 | |
| }, | |
| { | |
| "epoch": 0.5131065253764641, | |
| "grad_norm": 0.37761184573173523, | |
| "learning_rate": 0.00017071067811865476, | |
| "loss": 0.3312467098236084, | |
| "step": 230 | |
| }, | |
| { | |
| "epoch": 0.5354155047406581, | |
| "grad_norm": 0.5092398524284363, | |
| "learning_rate": 0.0001681884382763505, | |
| "loss": 0.3050385475158691, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.5577244841048522, | |
| "grad_norm": 0.4748888313770294, | |
| "learning_rate": 0.00016558237798618245, | |
| "loss": 0.3287826061248779, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.5800334634690463, | |
| "grad_norm": 0.5076217651367188, | |
| "learning_rate": 0.00016289570074050493, | |
| "loss": 0.35127427577972414, | |
| "step": 260 | |
| }, | |
| { | |
| "epoch": 0.6023424428332403, | |
| "grad_norm": 0.42200589179992676, | |
| "learning_rate": 0.00016013170912984058, | |
| "loss": 0.3230970144271851, | |
| "step": 270 | |
| }, | |
| { | |
| "epoch": 0.6246514221974344, | |
| "grad_norm": 0.5083448886871338, | |
| "learning_rate": 0.0001572938007831798, | |
| "loss": 0.3022065877914429, | |
| "step": 280 | |
| }, | |
| { | |
| "epoch": 0.6469604015616286, | |
| "grad_norm": 0.45063674449920654, | |
| "learning_rate": 0.00015438546419145488, | |
| "loss": 0.353403902053833, | |
| "step": 290 | |
| }, | |
| { | |
| "epoch": 0.6692693809258227, | |
| "grad_norm": 0.3552491366863251, | |
| "learning_rate": 0.00015141027441932216, | |
| "loss": 0.3080631494522095, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.6692693809258227, | |
| "eval_loss": 5.376068592071533, | |
| "eval_runtime": 187.8009, | |
| "eval_samples_per_second": 2.401, | |
| "eval_steps_per_second": 2.401, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.6915783602900167, | |
| "grad_norm": 0.4308546483516693, | |
| "learning_rate": 0.000148371888710524, | |
| "loss": 0.3246779918670654, | |
| "step": 310 | |
| }, | |
| { | |
| "epoch": 0.7138873396542108, | |
| "grad_norm": 0.6152803301811218, | |
| "learning_rate": 0.00014527404199223172, | |
| "loss": 0.35356783866882324, | |
| "step": 320 | |
| }, | |
| { | |
| "epoch": 0.7361963190184049, | |
| "grad_norm": 0.5230463147163391, | |
| "learning_rate": 0.0001421205422838971, | |
| "loss": 0.3224280834197998, | |
| "step": 330 | |
| }, | |
| { | |
| "epoch": 0.758505298382599, | |
| "grad_norm": 0.6994717121124268, | |
| "learning_rate": 0.0001389152660162549, | |
| "loss": 0.32481276988983154, | |
| "step": 340 | |
| }, | |
| { | |
| "epoch": 0.7808142777467931, | |
| "grad_norm": 0.5130417943000793, | |
| "learning_rate": 0.0001356621532662313, | |
| "loss": 0.28367717266082765, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.8031232571109872, | |
| "grad_norm": 0.59247887134552, | |
| "learning_rate": 0.00013236520291361515, | |
| "loss": 0.3666529655456543, | |
| "step": 360 | |
| }, | |
| { | |
| "epoch": 0.8254322364751813, | |
| "grad_norm": 0.3807213306427002, | |
| "learning_rate": 0.00012902846772544624, | |
| "loss": 0.3158045530319214, | |
| "step": 370 | |
| }, | |
| { | |
| "epoch": 0.8477412158393753, | |
| "grad_norm": 0.49467733502388, | |
| "learning_rate": 0.00012565604937416267, | |
| "loss": 0.2739120006561279, | |
| "step": 380 | |
| }, | |
| { | |
| "epoch": 0.8700501952035694, | |
| "grad_norm": 0.3233357071876526, | |
| "learning_rate": 0.00012225209339563145, | |
| "loss": 0.29424002170562746, | |
| "step": 390 | |
| }, | |
| { | |
| "epoch": 0.8923591745677635, | |
| "grad_norm": 0.5100898146629333, | |
| "learning_rate": 0.00011882078409326002, | |
| "loss": 0.3178868293762207, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.8923591745677635, | |
| "eval_loss": 5.181941986083984, | |
| "eval_runtime": 187.4258, | |
| "eval_samples_per_second": 2.406, | |
| "eval_steps_per_second": 2.406, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.9146681539319577, | |
| "grad_norm": 0.3522528409957886, | |
| "learning_rate": 0.000115366339394453, | |
| "loss": 0.3059010744094849, | |
| "step": 410 | |
| }, | |
| { | |
| "epoch": 0.9369771332961517, | |
| "grad_norm": 0.3386065661907196, | |
| "learning_rate": 0.0001118930056657367, | |
| "loss": 0.29911515712738035, | |
| "step": 420 | |
| }, | |
| { | |
| "epoch": 0.9592861126603458, | |
| "grad_norm": 0.4055585563182831, | |
| "learning_rate": 0.00010840505249292476, | |
| "loss": 0.266009521484375, | |
| "step": 430 | |
| }, | |
| { | |
| "epoch": 0.9815950920245399, | |
| "grad_norm": 0.6631172299385071, | |
| "learning_rate": 0.00010490676743274181, | |
| "loss": 0.28707849979400635, | |
| "step": 440 | |
| }, | |
| { | |
| "epoch": 1.0022308979364194, | |
| "grad_norm": 0.4735918641090393, | |
| "learning_rate": 0.00010140245074235624, | |
| "loss": 0.3470479726791382, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 1.0245398773006136, | |
| "grad_norm": 0.4347779452800751, | |
| "learning_rate": 9.789641009330111e-05, | |
| "loss": 0.2632521867752075, | |
| "step": 460 | |
| }, | |
| { | |
| "epoch": 1.0468488566648075, | |
| "grad_norm": 0.3260243833065033, | |
| "learning_rate": 9.439295527628081e-05, | |
| "loss": 0.2586169958114624, | |
| "step": 470 | |
| }, | |
| { | |
| "epoch": 1.0691578360290017, | |
| "grad_norm": 0.4574776291847229, | |
| "learning_rate": 9.0896392903373e-05, | |
| "loss": 0.2648995637893677, | |
| "step": 480 | |
| }, | |
| { | |
| "epoch": 1.0914668153931957, | |
| "grad_norm": 0.4806566834449768, | |
| "learning_rate": 8.741102111413748e-05, | |
| "loss": 0.24064080715179442, | |
| "step": 490 | |
| }, | |
| { | |
| "epoch": 1.1137757947573899, | |
| "grad_norm": 0.4908582270145416, | |
| "learning_rate": 8.39411242921403e-05, | |
| "loss": 0.25490708351135255, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 1.1137757947573899, | |
| "eval_loss": 5.465029239654541, | |
| "eval_runtime": 190.4691, | |
| "eval_samples_per_second": 2.368, | |
| "eval_steps_per_second": 2.368, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 1.136084774121584, | |
| "grad_norm": 0.4439888894557953, | |
| "learning_rate": 8.049096779838719e-05, | |
| "loss": 0.2863529920578003, | |
| "step": 510 | |
| }, | |
| { | |
| "epoch": 1.158393753485778, | |
| "grad_norm": 0.43033894896507263, | |
| "learning_rate": 7.706479272814023e-05, | |
| "loss": 0.2422431468963623, | |
| "step": 520 | |
| }, | |
| { | |
| "epoch": 1.1807027328499722, | |
| "grad_norm": 0.44316694140434265, | |
| "learning_rate": 7.366681069756352e-05, | |
| "loss": 0.27856547832489015, | |
| "step": 530 | |
| }, | |
| { | |
| "epoch": 1.2030117122141661, | |
| "grad_norm": 0.5816928744316101, | |
| "learning_rate": 7.030119866660564e-05, | |
| "loss": 0.26144585609436033, | |
| "step": 540 | |
| }, | |
| { | |
| "epoch": 1.2253206915783603, | |
| "grad_norm": 0.383112370967865, | |
| "learning_rate": 6.697209380448333e-05, | |
| "loss": 0.2592925548553467, | |
| "step": 550 | |
| }, | |
| { | |
| "epoch": 1.2476296709425543, | |
| "grad_norm": 0.4637056589126587, | |
| "learning_rate": 6.368358840407753e-05, | |
| "loss": 0.2502119779586792, | |
| "step": 560 | |
| }, | |
| { | |
| "epoch": 1.2699386503067485, | |
| "grad_norm": 0.3978555202484131, | |
| "learning_rate": 6.043972485149414e-05, | |
| "loss": 0.24019856452941896, | |
| "step": 570 | |
| }, | |
| { | |
| "epoch": 1.2922476296709426, | |
| "grad_norm": 0.3794662654399872, | |
| "learning_rate": 5.7244490656971815e-05, | |
| "loss": 0.23192195892333983, | |
| "step": 580 | |
| }, | |
| { | |
| "epoch": 1.3145566090351366, | |
| "grad_norm": 0.34926116466522217, | |
| "learning_rate": 5.410181355324622e-05, | |
| "loss": 0.26394219398498536, | |
| "step": 590 | |
| }, | |
| { | |
| "epoch": 1.3368655883993308, | |
| "grad_norm": 0.4157903492450714, | |
| "learning_rate": 5.1015556667395636e-05, | |
| "loss": 0.2660749197006226, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 1.3368655883993308, | |
| "eval_loss": 5.384934425354004, | |
| "eval_runtime": 188.6552, | |
| "eval_samples_per_second": 2.391, | |
| "eval_steps_per_second": 2.391, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 1.3591745677635247, | |
| "grad_norm": 0.5136097073554993, | |
| "learning_rate": 4.7989513772102537e-05, | |
| "loss": 0.23524351119995118, | |
| "step": 610 | |
| }, | |
| { | |
| "epoch": 1.381483547127719, | |
| "grad_norm": 0.3920697867870331, | |
| "learning_rate": 4.502740462216919e-05, | |
| "loss": 0.2516023635864258, | |
| "step": 620 | |
| }, | |
| { | |
| "epoch": 1.4037925264919129, | |
| "grad_norm": 0.5687846541404724, | |
| "learning_rate": 4.213287038201943e-05, | |
| "loss": 0.23717131614685058, | |
| "step": 630 | |
| }, | |
| { | |
| "epoch": 1.426101505856107, | |
| "grad_norm": 0.8705297112464905, | |
| "learning_rate": 3.930946914980744e-05, | |
| "loss": 0.22555122375488282, | |
| "step": 640 | |
| }, | |
| { | |
| "epoch": 1.4484104852203012, | |
| "grad_norm": 0.4317159652709961, | |
| "learning_rate": 3.6560671583635467e-05, | |
| "loss": 0.2330761432647705, | |
| "step": 650 | |
| }, | |
| { | |
| "epoch": 1.4707194645844952, | |
| "grad_norm": 0.2786170244216919, | |
| "learning_rate": 3.388985663525702e-05, | |
| "loss": 0.24024245738983155, | |
| "step": 660 | |
| }, | |
| { | |
| "epoch": 1.4930284439486894, | |
| "grad_norm": 0.5246191620826721, | |
| "learning_rate": 3.130030739650983e-05, | |
| "loss": 0.21560235023498536, | |
| "step": 670 | |
| }, | |
| { | |
| "epoch": 1.5153374233128836, | |
| "grad_norm": 0.3500918745994568, | |
| "learning_rate": 2.879520706358446e-05, | |
| "loss": 0.26663081645965575, | |
| "step": 680 | |
| }, | |
| { | |
| "epoch": 1.5376464026770775, | |
| "grad_norm": 0.44421038031578064, | |
| "learning_rate": 2.6377635024089087e-05, | |
| "loss": 0.28586716651916505, | |
| "step": 690 | |
| }, | |
| { | |
| "epoch": 1.5599553820412715, | |
| "grad_norm": 0.5229424238204956, | |
| "learning_rate": 2.4050563071720867e-05, | |
| "loss": 0.2517236709594727, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 1.5599553820412715, | |
| "eval_loss": 5.481088161468506, | |
| "eval_runtime": 189.8684, | |
| "eval_samples_per_second": 2.375, | |
| "eval_steps_per_second": 2.375, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 1.5822643614054657, | |
| "grad_norm": 0.4634057879447937, | |
| "learning_rate": 2.181685175319702e-05, | |
| "loss": 0.2557069778442383, | |
| "step": 710 | |
| }, | |
| { | |
| "epoch": 1.6045733407696599, | |
| "grad_norm": 0.4304831624031067, | |
| "learning_rate": 1.967924685193552e-05, | |
| "loss": 0.2168825626373291, | |
| "step": 720 | |
| }, | |
| { | |
| "epoch": 1.6268823201338538, | |
| "grad_norm": 0.4848768711090088, | |
| "learning_rate": 1.7640376012808536e-05, | |
| "loss": 0.238733172416687, | |
| "step": 730 | |
| }, | |
| { | |
| "epoch": 1.649191299498048, | |
| "grad_norm": 0.5546223521232605, | |
| "learning_rate": 1.5702745512117324e-05, | |
| "loss": 0.26332356929779055, | |
| "step": 740 | |
| }, | |
| { | |
| "epoch": 1.6715002788622422, | |
| "grad_norm": 0.4730488955974579, | |
| "learning_rate": 1.3868737176759106e-05, | |
| "loss": 0.25144917964935304, | |
| "step": 750 | |
| }, | |
| { | |
| "epoch": 1.6938092582264361, | |
| "grad_norm": 0.5579213500022888, | |
| "learning_rate": 1.2140605456372855e-05, | |
| "loss": 0.26153883934020994, | |
| "step": 760 | |
| }, | |
| { | |
| "epoch": 1.71611823759063, | |
| "grad_norm": 0.5828114748001099, | |
| "learning_rate": 1.0520474652063394e-05, | |
| "loss": 0.2530355930328369, | |
| "step": 770 | |
| }, | |
| { | |
| "epoch": 1.7384272169548243, | |
| "grad_norm": 0.38401705026626587, | |
| "learning_rate": 9.010336305110345e-06, | |
| "loss": 0.24105710983276368, | |
| "step": 780 | |
| }, | |
| { | |
| "epoch": 1.7607361963190185, | |
| "grad_norm": 0.6085273027420044, | |
| "learning_rate": 7.612046748871327e-06, | |
| "loss": 0.24265129566192628, | |
| "step": 790 | |
| }, | |
| { | |
| "epoch": 1.7830451756832124, | |
| "grad_norm": 0.3713201582431793, | |
| "learning_rate": 6.327324826889469e-06, | |
| "loss": 0.2452718734741211, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 1.7830451756832124, | |
| "eval_loss": 5.595068454742432, | |
| "eval_runtime": 188.2887, | |
| "eval_samples_per_second": 2.395, | |
| "eval_steps_per_second": 2.395, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 1.8053541550474066, | |
| "grad_norm": 0.3409222364425659, | |
| "learning_rate": 5.157749780009735e-06, | |
| "loss": 0.22725944519042968, | |
| "step": 810 | |
| }, | |
| { | |
| "epoch": 1.8276631344116008, | |
| "grad_norm": 0.5552815794944763, | |
| "learning_rate": 4.104759305101525e-06, | |
| "loss": 0.22626399993896484, | |
| "step": 820 | |
| }, | |
| { | |
| "epoch": 1.8499721137757947, | |
| "grad_norm": 0.561656653881073, | |
| "learning_rate": 3.169647787773866e-06, | |
| "loss": 0.2625704526901245, | |
| "step": 830 | |
| }, | |
| { | |
| "epoch": 1.8722810931399887, | |
| "grad_norm": 0.5719916820526123, | |
| "learning_rate": 2.3535647112553294e-06, | |
| "loss": 0.2593970775604248, | |
| "step": 840 | |
| }, | |
| { | |
| "epoch": 1.894590072504183, | |
| "grad_norm": 0.41392213106155396, | |
| "learning_rate": 1.657513243395159e-06, | |
| "loss": 0.22938990592956543, | |
| "step": 850 | |
| } | |
| ], | |
| "logging_steps": 10, | |
| "max_steps": 898, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 2, | |
| "save_steps": 50, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1.367701442172864e+16, | |
| "train_batch_size": 1, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |