Unconditional Image Generation
Transformers
Safetensors
tinyimagegen
feature-extraction
imagegen
unconditional-image
custom_code
Instructions to use fromziro/TinyImageGen-0.6M with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use fromziro/TinyImageGen-0.6M with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("fromziro/TinyImageGen-0.6M", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 15.0, | |
| "eval_steps": 500, | |
| "global_step": 5865, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.1278772378516624, | |
| "grad_norm": 1.4250309467315674, | |
| "learning_rate": 0.00016723549488054606, | |
| "loss": 1.3987660217285156, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.2557544757033248, | |
| "grad_norm": 0.8242300748825073, | |
| "learning_rate": 0.0003378839590443686, | |
| "loss": 0.8584999847412109, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.3836317135549872, | |
| "grad_norm": 1.9371459484100342, | |
| "learning_rate": 0.0005085324232081912, | |
| "loss": 0.4975930786132812, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.5115089514066496, | |
| "grad_norm": 1.5103962421417236, | |
| "learning_rate": 0.0006791808873720137, | |
| "loss": 0.4130295944213867, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.639386189258312, | |
| "grad_norm": 1.4034031629562378, | |
| "learning_rate": 0.0008498293515358362, | |
| "loss": 0.35486129760742186, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.7672634271099744, | |
| "grad_norm": 2.0146305561065674, | |
| "learning_rate": 0.001, | |
| "loss": 0.3292802047729492, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.8951406649616368, | |
| "grad_norm": 0.965383768081665, | |
| "learning_rate": 0.001, | |
| "loss": 0.3188431739807129, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 1.0230179028132993, | |
| "grad_norm": 1.1068569421768188, | |
| "learning_rate": 0.001, | |
| "loss": 0.30395978927612305, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 1.1508951406649617, | |
| "grad_norm": 1.4761624336242676, | |
| "learning_rate": 0.001, | |
| "loss": 0.29518972396850585, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 1.278772378516624, | |
| "grad_norm": 0.7133429050445557, | |
| "learning_rate": 0.001, | |
| "loss": 0.27936880111694334, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 1.278772378516624, | |
| "eval_runtime": 1.25, | |
| "eval_samples_per_second": 39.999, | |
| "eval_steps_per_second": 0.8, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 1.4066496163682864, | |
| "grad_norm": 0.588685929775238, | |
| "learning_rate": 0.001, | |
| "loss": 0.2743060111999512, | |
| "step": 550 | |
| }, | |
| { | |
| "epoch": 1.5345268542199488, | |
| "grad_norm": 0.8678897619247437, | |
| "learning_rate": 0.001, | |
| "loss": 0.27374002456665036, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 1.6624040920716112, | |
| "grad_norm": 0.6150850653648376, | |
| "learning_rate": 0.001, | |
| "loss": 0.2637761878967285, | |
| "step": 650 | |
| }, | |
| { | |
| "epoch": 1.7902813299232738, | |
| "grad_norm": 1.0410972833633423, | |
| "learning_rate": 0.001, | |
| "loss": 0.25924995422363284, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 1.918158567774936, | |
| "grad_norm": 0.5429728031158447, | |
| "learning_rate": 0.001, | |
| "loss": 0.2584137725830078, | |
| "step": 750 | |
| }, | |
| { | |
| "epoch": 2.0460358056265986, | |
| "grad_norm": 0.5936292409896851, | |
| "learning_rate": 0.001, | |
| "loss": 0.2569990348815918, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 2.1739130434782608, | |
| "grad_norm": 0.5811658501625061, | |
| "learning_rate": 0.001, | |
| "loss": 0.2502941131591797, | |
| "step": 850 | |
| }, | |
| { | |
| "epoch": 2.3017902813299234, | |
| "grad_norm": 0.5268040895462036, | |
| "learning_rate": 0.001, | |
| "loss": 0.24979625701904296, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 2.4296675191815855, | |
| "grad_norm": 0.5837516784667969, | |
| "learning_rate": 0.001, | |
| "loss": 0.24781970977783202, | |
| "step": 950 | |
| }, | |
| { | |
| "epoch": 2.557544757033248, | |
| "grad_norm": 0.3708310127258301, | |
| "learning_rate": 0.001, | |
| "loss": 0.24778980255126953, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 2.557544757033248, | |
| "eval_runtime": 0.5765, | |
| "eval_samples_per_second": 86.733, | |
| "eval_steps_per_second": 1.735, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 2.6854219948849103, | |
| "grad_norm": 0.573240339756012, | |
| "learning_rate": 0.001, | |
| "loss": 0.24511468887329102, | |
| "step": 1050 | |
| }, | |
| { | |
| "epoch": 2.813299232736573, | |
| "grad_norm": 0.47702646255493164, | |
| "learning_rate": 0.001, | |
| "loss": 0.24337112426757812, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 2.9411764705882355, | |
| "grad_norm": 0.5146952867507935, | |
| "learning_rate": 0.001, | |
| "loss": 0.24384737014770508, | |
| "step": 1150 | |
| }, | |
| { | |
| "epoch": 3.0690537084398977, | |
| "grad_norm": 0.5712032318115234, | |
| "learning_rate": 0.001, | |
| "loss": 0.241769962310791, | |
| "step": 1200 | |
| }, | |
| { | |
| "epoch": 3.1969309462915603, | |
| "grad_norm": 0.5123623013496399, | |
| "learning_rate": 0.001, | |
| "loss": 0.23703174591064452, | |
| "step": 1250 | |
| }, | |
| { | |
| "epoch": 3.3248081841432224, | |
| "grad_norm": 0.4334401786327362, | |
| "learning_rate": 0.001, | |
| "loss": 0.2368149185180664, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 3.452685421994885, | |
| "grad_norm": 0.30235975980758667, | |
| "learning_rate": 0.001, | |
| "loss": 0.2367188835144043, | |
| "step": 1350 | |
| }, | |
| { | |
| "epoch": 3.580562659846547, | |
| "grad_norm": 0.5382092595100403, | |
| "learning_rate": 0.001, | |
| "loss": 0.23868675231933595, | |
| "step": 1400 | |
| }, | |
| { | |
| "epoch": 3.70843989769821, | |
| "grad_norm": 0.6381360292434692, | |
| "learning_rate": 0.001, | |
| "loss": 0.2370777893066406, | |
| "step": 1450 | |
| }, | |
| { | |
| "epoch": 3.836317135549872, | |
| "grad_norm": 0.30094292759895325, | |
| "learning_rate": 0.001, | |
| "loss": 0.23615116119384766, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 3.836317135549872, | |
| "eval_runtime": 0.5583, | |
| "eval_samples_per_second": 89.562, | |
| "eval_steps_per_second": 1.791, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 3.9641943734015346, | |
| "grad_norm": 0.4271095395088196, | |
| "learning_rate": 0.001, | |
| "loss": 0.23447750091552735, | |
| "step": 1550 | |
| }, | |
| { | |
| "epoch": 4.092071611253197, | |
| "grad_norm": 0.5157049894332886, | |
| "learning_rate": 0.001, | |
| "loss": 0.23318645477294922, | |
| "step": 1600 | |
| }, | |
| { | |
| "epoch": 4.21994884910486, | |
| "grad_norm": 0.4829655587673187, | |
| "learning_rate": 0.001, | |
| "loss": 0.23322877883911133, | |
| "step": 1650 | |
| }, | |
| { | |
| "epoch": 4.3478260869565215, | |
| "grad_norm": 0.44096723198890686, | |
| "learning_rate": 0.001, | |
| "loss": 0.23009027481079103, | |
| "step": 1700 | |
| }, | |
| { | |
| "epoch": 4.475703324808184, | |
| "grad_norm": 0.3808368444442749, | |
| "learning_rate": 0.001, | |
| "loss": 0.23304725646972657, | |
| "step": 1750 | |
| }, | |
| { | |
| "epoch": 4.603580562659847, | |
| "grad_norm": 0.40833979845046997, | |
| "learning_rate": 0.001, | |
| "loss": 0.2287454605102539, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 4.731457800511509, | |
| "grad_norm": 0.5448183417320251, | |
| "learning_rate": 0.001, | |
| "loss": 0.23105720520019532, | |
| "step": 1850 | |
| }, | |
| { | |
| "epoch": 4.859335038363171, | |
| "grad_norm": 0.34567973017692566, | |
| "learning_rate": 0.001, | |
| "loss": 0.23205410003662108, | |
| "step": 1900 | |
| }, | |
| { | |
| "epoch": 4.987212276214834, | |
| "grad_norm": 0.41231873631477356, | |
| "learning_rate": 0.001, | |
| "loss": 0.22798387527465822, | |
| "step": 1950 | |
| }, | |
| { | |
| "epoch": 5.115089514066496, | |
| "grad_norm": 0.31789979338645935, | |
| "learning_rate": 0.001, | |
| "loss": 0.2274525260925293, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 5.115089514066496, | |
| "eval_runtime": 0.6168, | |
| "eval_samples_per_second": 81.069, | |
| "eval_steps_per_second": 1.621, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 5.242966751918159, | |
| "grad_norm": 0.35738807916641235, | |
| "learning_rate": 0.001, | |
| "loss": 0.23131420135498046, | |
| "step": 2050 | |
| }, | |
| { | |
| "epoch": 5.370843989769821, | |
| "grad_norm": 0.5555620193481445, | |
| "learning_rate": 0.001, | |
| "loss": 0.2268832778930664, | |
| "step": 2100 | |
| }, | |
| { | |
| "epoch": 5.498721227621483, | |
| "grad_norm": 0.3198654055595398, | |
| "learning_rate": 0.001, | |
| "loss": 0.22977170944213868, | |
| "step": 2150 | |
| }, | |
| { | |
| "epoch": 5.626598465473146, | |
| "grad_norm": 0.3937130868434906, | |
| "learning_rate": 0.001, | |
| "loss": 0.23172990798950197, | |
| "step": 2200 | |
| }, | |
| { | |
| "epoch": 5.754475703324808, | |
| "grad_norm": 0.37701380252838135, | |
| "learning_rate": 0.001, | |
| "loss": 0.22774370193481444, | |
| "step": 2250 | |
| }, | |
| { | |
| "epoch": 5.882352941176471, | |
| "grad_norm": 0.2647969722747803, | |
| "learning_rate": 0.001, | |
| "loss": 0.2257476806640625, | |
| "step": 2300 | |
| }, | |
| { | |
| "epoch": 6.010230179028133, | |
| "grad_norm": 0.298615962266922, | |
| "learning_rate": 0.001, | |
| "loss": 0.22943933486938475, | |
| "step": 2350 | |
| }, | |
| { | |
| "epoch": 6.138107416879795, | |
| "grad_norm": 0.36317694187164307, | |
| "learning_rate": 0.001, | |
| "loss": 0.2241596794128418, | |
| "step": 2400 | |
| }, | |
| { | |
| "epoch": 6.265984654731458, | |
| "grad_norm": 0.3090345561504364, | |
| "learning_rate": 0.001, | |
| "loss": 0.22605371475219727, | |
| "step": 2450 | |
| }, | |
| { | |
| "epoch": 6.3938618925831205, | |
| "grad_norm": 0.3888801634311676, | |
| "learning_rate": 0.001, | |
| "loss": 0.225456600189209, | |
| "step": 2500 | |
| }, | |
| { | |
| "epoch": 6.3938618925831205, | |
| "eval_runtime": 0.6059, | |
| "eval_samples_per_second": 82.522, | |
| "eval_steps_per_second": 1.65, | |
| "step": 2500 | |
| }, | |
| { | |
| "epoch": 6.521739130434782, | |
| "grad_norm": 0.41455891728401184, | |
| "learning_rate": 0.001, | |
| "loss": 0.22721860885620118, | |
| "step": 2550 | |
| }, | |
| { | |
| "epoch": 6.649616368286445, | |
| "grad_norm": 0.4250830113887787, | |
| "learning_rate": 0.001, | |
| "loss": 0.22337587356567382, | |
| "step": 2600 | |
| }, | |
| { | |
| "epoch": 6.7774936061381075, | |
| "grad_norm": 0.42290645837783813, | |
| "learning_rate": 0.001, | |
| "loss": 0.2271786880493164, | |
| "step": 2650 | |
| }, | |
| { | |
| "epoch": 6.90537084398977, | |
| "grad_norm": 0.2795289158821106, | |
| "learning_rate": 0.001, | |
| "loss": 0.22438146591186522, | |
| "step": 2700 | |
| }, | |
| { | |
| "epoch": 7.033248081841432, | |
| "grad_norm": 0.29143092036247253, | |
| "learning_rate": 0.001, | |
| "loss": 0.2258744239807129, | |
| "step": 2750 | |
| }, | |
| { | |
| "epoch": 7.161125319693094, | |
| "grad_norm": 0.5103092789649963, | |
| "learning_rate": 0.001, | |
| "loss": 0.2245882797241211, | |
| "step": 2800 | |
| }, | |
| { | |
| "epoch": 7.289002557544757, | |
| "grad_norm": 0.45546701550483704, | |
| "learning_rate": 0.001, | |
| "loss": 0.2228009033203125, | |
| "step": 2850 | |
| }, | |
| { | |
| "epoch": 7.41687979539642, | |
| "grad_norm": 0.30181851983070374, | |
| "learning_rate": 0.001, | |
| "loss": 0.22455228805541994, | |
| "step": 2900 | |
| }, | |
| { | |
| "epoch": 7.544757033248082, | |
| "grad_norm": 0.5398461222648621, | |
| "learning_rate": 0.001, | |
| "loss": 0.22371641159057618, | |
| "step": 2950 | |
| }, | |
| { | |
| "epoch": 7.672634271099744, | |
| "grad_norm": 0.2513619363307953, | |
| "learning_rate": 0.001, | |
| "loss": 0.22418743133544922, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 7.672634271099744, | |
| "eval_runtime": 0.5955, | |
| "eval_samples_per_second": 83.968, | |
| "eval_steps_per_second": 1.679, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 7.8005115089514065, | |
| "grad_norm": 0.45771321654319763, | |
| "learning_rate": 0.001, | |
| "loss": 0.22467132568359374, | |
| "step": 3050 | |
| }, | |
| { | |
| "epoch": 7.928388746803069, | |
| "grad_norm": 0.32969486713409424, | |
| "learning_rate": 0.001, | |
| "loss": 0.22461874008178712, | |
| "step": 3100 | |
| }, | |
| { | |
| "epoch": 8.05626598465473, | |
| "grad_norm": 0.26984792947769165, | |
| "learning_rate": 0.001, | |
| "loss": 0.2249747085571289, | |
| "step": 3150 | |
| }, | |
| { | |
| "epoch": 8.184143222506394, | |
| "grad_norm": 0.3755267560482025, | |
| "learning_rate": 0.001, | |
| "loss": 0.22293462753295898, | |
| "step": 3200 | |
| }, | |
| { | |
| "epoch": 8.312020460358056, | |
| "grad_norm": 0.3465856909751892, | |
| "learning_rate": 0.001, | |
| "loss": 0.22129749298095702, | |
| "step": 3250 | |
| }, | |
| { | |
| "epoch": 8.43989769820972, | |
| "grad_norm": 0.4587877690792084, | |
| "learning_rate": 0.001, | |
| "loss": 0.22289451599121093, | |
| "step": 3300 | |
| }, | |
| { | |
| "epoch": 8.567774936061381, | |
| "grad_norm": 0.3278965353965759, | |
| "learning_rate": 0.001, | |
| "loss": 0.21755035400390624, | |
| "step": 3350 | |
| }, | |
| { | |
| "epoch": 8.695652173913043, | |
| "grad_norm": 0.4856521785259247, | |
| "learning_rate": 0.001, | |
| "loss": 0.22219255447387695, | |
| "step": 3400 | |
| }, | |
| { | |
| "epoch": 8.823529411764707, | |
| "grad_norm": 0.24720615148544312, | |
| "learning_rate": 0.001, | |
| "loss": 0.22077470779418945, | |
| "step": 3450 | |
| }, | |
| { | |
| "epoch": 8.951406649616368, | |
| "grad_norm": 0.3050568699836731, | |
| "learning_rate": 0.001, | |
| "loss": 0.22129110336303712, | |
| "step": 3500 | |
| }, | |
| { | |
| "epoch": 8.951406649616368, | |
| "eval_runtime": 0.5745, | |
| "eval_samples_per_second": 87.031, | |
| "eval_steps_per_second": 1.741, | |
| "step": 3500 | |
| }, | |
| { | |
| "epoch": 9.07928388746803, | |
| "grad_norm": 0.19966809451580048, | |
| "learning_rate": 0.001, | |
| "loss": 0.2235894775390625, | |
| "step": 3550 | |
| }, | |
| { | |
| "epoch": 9.207161125319693, | |
| "grad_norm": 0.3338870406150818, | |
| "learning_rate": 0.001, | |
| "loss": 0.22133295059204103, | |
| "step": 3600 | |
| }, | |
| { | |
| "epoch": 9.335038363171355, | |
| "grad_norm": 0.2079731523990631, | |
| "learning_rate": 0.001, | |
| "loss": 0.22170833587646485, | |
| "step": 3650 | |
| }, | |
| { | |
| "epoch": 9.462915601023019, | |
| "grad_norm": 0.30865564942359924, | |
| "learning_rate": 0.001, | |
| "loss": 0.22252700805664063, | |
| "step": 3700 | |
| }, | |
| { | |
| "epoch": 9.59079283887468, | |
| "grad_norm": 0.38541874289512634, | |
| "learning_rate": 0.001, | |
| "loss": 0.22342273712158203, | |
| "step": 3750 | |
| }, | |
| { | |
| "epoch": 9.718670076726342, | |
| "grad_norm": 0.2266816347837448, | |
| "learning_rate": 0.001, | |
| "loss": 0.22231040954589842, | |
| "step": 3800 | |
| }, | |
| { | |
| "epoch": 9.846547314578006, | |
| "grad_norm": 0.3022933304309845, | |
| "learning_rate": 0.001, | |
| "loss": 0.22152088165283204, | |
| "step": 3850 | |
| }, | |
| { | |
| "epoch": 9.974424552429667, | |
| "grad_norm": 0.31352734565734863, | |
| "learning_rate": 0.001, | |
| "loss": 0.22159164428710937, | |
| "step": 3900 | |
| }, | |
| { | |
| "epoch": 10.10230179028133, | |
| "grad_norm": 0.292133092880249, | |
| "learning_rate": 0.001, | |
| "loss": 0.22279462814331055, | |
| "step": 3950 | |
| }, | |
| { | |
| "epoch": 10.230179028132993, | |
| "grad_norm": 0.3261716663837433, | |
| "learning_rate": 0.001, | |
| "loss": 0.2177202033996582, | |
| "step": 4000 | |
| }, | |
| { | |
| "epoch": 10.230179028132993, | |
| "eval_runtime": 0.5615, | |
| "eval_samples_per_second": 89.043, | |
| "eval_steps_per_second": 1.781, | |
| "step": 4000 | |
| }, | |
| { | |
| "epoch": 10.358056265984654, | |
| "grad_norm": 0.5240085124969482, | |
| "learning_rate": 0.001, | |
| "loss": 0.21939998626708984, | |
| "step": 4050 | |
| }, | |
| { | |
| "epoch": 10.485933503836318, | |
| "grad_norm": 0.28354591131210327, | |
| "learning_rate": 0.001, | |
| "loss": 0.21681865692138672, | |
| "step": 4100 | |
| }, | |
| { | |
| "epoch": 10.61381074168798, | |
| "grad_norm": 0.3266175389289856, | |
| "learning_rate": 0.000998458666866564, | |
| "loss": 0.22585365295410156, | |
| "step": 4150 | |
| }, | |
| { | |
| "epoch": 10.741687979539641, | |
| "grad_norm": 0.9483559131622314, | |
| "learning_rate": 0.0009929781591383633, | |
| "loss": 0.2212784194946289, | |
| "step": 4200 | |
| }, | |
| { | |
| "epoch": 10.869565217391305, | |
| "grad_norm": 0.25296592712402344, | |
| "learning_rate": 0.0009835734273509786, | |
| "loss": 0.219591007232666, | |
| "step": 4250 | |
| }, | |
| { | |
| "epoch": 10.997442455242966, | |
| "grad_norm": 0.23266983032226562, | |
| "learning_rate": 0.0009703193354188977, | |
| "loss": 0.22084575653076172, | |
| "step": 4300 | |
| }, | |
| { | |
| "epoch": 11.12531969309463, | |
| "grad_norm": 0.31800100207328796, | |
| "learning_rate": 0.0009533213890840658, | |
| "loss": 0.2211077880859375, | |
| "step": 4350 | |
| }, | |
| { | |
| "epoch": 11.253196930946292, | |
| "grad_norm": 0.2673555314540863, | |
| "learning_rate": 0.0009327148960649431, | |
| "loss": 0.21858774185180663, | |
| "step": 4400 | |
| }, | |
| { | |
| "epoch": 11.381074168797953, | |
| "grad_norm": 0.36419302225112915, | |
| "learning_rate": 0.0009086638889747034, | |
| "loss": 0.2180442237854004, | |
| "step": 4450 | |
| }, | |
| { | |
| "epoch": 11.508951406649617, | |
| "grad_norm": 0.2488548904657364, | |
| "learning_rate": 0.0008813598195823991, | |
| "loss": 0.21677488327026367, | |
| "step": 4500 | |
| }, | |
| { | |
| "epoch": 11.508951406649617, | |
| "eval_runtime": 0.6061, | |
| "eval_samples_per_second": 82.497, | |
| "eval_steps_per_second": 1.65, | |
| "step": 4500 | |
| }, | |
| { | |
| "epoch": 11.636828644501279, | |
| "grad_norm": 0.1898481398820877, | |
| "learning_rate": 0.0008510200348110868, | |
| "loss": 0.21867218017578124, | |
| "step": 4550 | |
| }, | |
| { | |
| "epoch": 11.764705882352942, | |
| "grad_norm": 0.26611390709877014, | |
| "learning_rate": 0.0008178860466043344, | |
| "loss": 0.218858642578125, | |
| "step": 4600 | |
| }, | |
| { | |
| "epoch": 11.892583120204604, | |
| "grad_norm": 0.24872389435768127, | |
| "learning_rate": 0.0007822216094333848, | |
| "loss": 0.21927881240844727, | |
| "step": 4650 | |
| }, | |
| { | |
| "epoch": 12.020460358056265, | |
| "grad_norm": 0.3024843633174896, | |
| "learning_rate": 0.0007443106207484776, | |
| "loss": 0.21875694274902344, | |
| "step": 4700 | |
| }, | |
| { | |
| "epoch": 12.148337595907929, | |
| "grad_norm": 0.25236737728118896, | |
| "learning_rate": 0.0007044548610872435, | |
| "loss": 0.21932044982910157, | |
| "step": 4750 | |
| }, | |
| { | |
| "epoch": 12.27621483375959, | |
| "grad_norm": 0.22667436301708221, | |
| "learning_rate": 0.0006629715918294422, | |
| "loss": 0.22077136993408203, | |
| "step": 4800 | |
| }, | |
| { | |
| "epoch": 12.404092071611252, | |
| "grad_norm": 0.2894568145275116, | |
| "learning_rate": 0.0006201910297204962, | |
| "loss": 0.22006725311279296, | |
| "step": 4850 | |
| }, | |
| { | |
| "epoch": 12.531969309462916, | |
| "grad_norm": 0.2752484381198883, | |
| "learning_rate": 0.000576453718267215, | |
| "loss": 0.21579278945922853, | |
| "step": 4900 | |
| }, | |
| { | |
| "epoch": 12.659846547314578, | |
| "grad_norm": 0.2868839502334595, | |
| "learning_rate": 0.0005321078169300275, | |
| "loss": 0.21556739807128905, | |
| "step": 4950 | |
| }, | |
| { | |
| "epoch": 12.787723785166241, | |
| "grad_norm": 0.1955188363790512, | |
| "learning_rate": 0.0004875063296904041, | |
| "loss": 0.2185918426513672, | |
| "step": 5000 | |
| }, | |
| { | |
| "epoch": 12.787723785166241, | |
| "eval_runtime": 0.6016, | |
| "eval_samples_per_second": 83.116, | |
| "eval_steps_per_second": 1.662, | |
| "step": 5000 | |
| }, | |
| { | |
| "epoch": 12.915601023017903, | |
| "grad_norm": 0.2818523943424225, | |
| "learning_rate": 0.0004430042950547297, | |
| "loss": 0.2146390914916992, | |
| "step": 5050 | |
| }, | |
| { | |
| "epoch": 13.043478260869565, | |
| "grad_norm": 0.20297877490520477, | |
| "learning_rate": 0.00039895595986287526, | |
| "loss": 0.21721738815307617, | |
| "step": 5100 | |
| }, | |
| { | |
| "epoch": 13.171355498721228, | |
| "grad_norm": 0.2772619128227234, | |
| "learning_rate": 0.00035571195939862075, | |
| "loss": 0.21498830795288085, | |
| "step": 5150 | |
| }, | |
| { | |
| "epoch": 13.29923273657289, | |
| "grad_norm": 0.34799399971961975, | |
| "learning_rate": 0.0003136165262489306, | |
| "loss": 0.21611639022827148, | |
| "step": 5200 | |
| }, | |
| { | |
| "epoch": 13.427109974424553, | |
| "grad_norm": 0.19180037081241608, | |
| "learning_rate": 0.00027300475013022663, | |
| "loss": 0.2164902687072754, | |
| "step": 5250 | |
| }, | |
| { | |
| "epoch": 13.554987212276215, | |
| "grad_norm": 0.3067183792591095, | |
| "learning_rate": 0.00023419991049409712, | |
| "loss": 0.21705455780029298, | |
| "step": 5300 | |
| }, | |
| { | |
| "epoch": 13.682864450127877, | |
| "grad_norm": 0.29999950528144836, | |
| "learning_rate": 0.00019751090314553877, | |
| "loss": 0.2135419464111328, | |
| "step": 5350 | |
| }, | |
| { | |
| "epoch": 13.81074168797954, | |
| "grad_norm": 0.15418510138988495, | |
| "learning_rate": 0.00016322978135846627, | |
| "loss": 0.2127900505065918, | |
| "step": 5400 | |
| }, | |
| { | |
| "epoch": 13.938618925831202, | |
| "grad_norm": 0.15595944225788116, | |
| "learning_rate": 0.00013162943106179747, | |
| "loss": 0.2144439125061035, | |
| "step": 5450 | |
| }, | |
| { | |
| "epoch": 14.066496163682864, | |
| "grad_norm": 0.1308145970106125, | |
| "learning_rate": 0.00010296139860218928, | |
| "loss": 0.21586280822753906, | |
| "step": 5500 | |
| }, | |
| { | |
| "epoch": 14.066496163682864, | |
| "eval_runtime": 0.6049, | |
| "eval_samples_per_second": 82.653, | |
| "eval_steps_per_second": 1.653, | |
| "step": 5500 | |
| }, | |
| { | |
| "epoch": 14.194373401534527, | |
| "grad_norm": 0.1245792955160141, | |
| "learning_rate": 7.745388837495188e-05, | |
| "loss": 0.21093259811401366, | |
| "step": 5550 | |
| }, | |
| { | |
| "epoch": 14.322250639386189, | |
| "grad_norm": 0.2582224905490875, | |
| "learning_rate": 5.530994626248026e-05, | |
| "loss": 0.2127166748046875, | |
| "step": 5600 | |
| }, | |
| { | |
| "epoch": 14.450127877237852, | |
| "grad_norm": 0.23266255855560303, | |
| "learning_rate": 3.6705843340464284e-05, | |
| "loss": 0.21203529357910156, | |
| "step": 5650 | |
| }, | |
| { | |
| "epoch": 14.578005115089514, | |
| "grad_norm": 0.14757439494132996, | |
| "learning_rate": 2.1789672717965227e-05, | |
| "loss": 0.2155706787109375, | |
| "step": 5700 | |
| }, | |
| { | |
| "epoch": 14.705882352941176, | |
| "grad_norm": 0.2186284065246582, | |
| "learning_rate": 1.0680170680846258e-05, | |
| "loss": 0.21344688415527344, | |
| "step": 5750 | |
| }, | |
| { | |
| "epoch": 14.83375959079284, | |
| "grad_norm": 0.09984402358531952, | |
| "learning_rate": 3.4657715225368535e-06, | |
| "loss": 0.21246673583984374, | |
| "step": 5800 | |
| }, | |
| { | |
| "epoch": 14.961636828644501, | |
| "grad_norm": 0.13794586062431335, | |
| "learning_rate": 2.0390358590538505e-07, | |
| "loss": 0.21189260482788086, | |
| "step": 5850 | |
| }, | |
| { | |
| "epoch": 15.0, | |
| "eval_runtime": 0.5425, | |
| "eval_samples_per_second": 92.167, | |
| "eval_steps_per_second": 1.843, | |
| "step": 5865 | |
| } | |
| ], | |
| "logging_steps": 50, | |
| "max_steps": 5865, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 15, | |
| "save_steps": 500, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 0.0, | |
| "train_batch_size": 128, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |