{ "HuggingFaceTB/SmolLM2-360M": { "star_name": "SmolLM2-360M", "title": "Stellar Beacon of SmolLM", "motto": "Lightweight wisdom, boundless knowledge", "bio": "SmolLM2-360M, a 361.82-million-parameter model, excels in instruction following, reasoning, and knowledge tasks. Trained on 4 trillion tokens from FineWeb-Edu, DCLM, The Stack, and curated datasets, it remains lightweight enough for on-device deployment. Its instruct version, fine-tuned with UltraFeedback, supports rewriting, summarization, and function calling.", "author_bio": "Hugging Face, a leading AI research and deployment organization, created and maintains the SmolLM2 family.", "specialty": "Counterfactual", "spectral_class": "G", "constellation": "The HuggingFaceTB Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffe9a8", "links": { "repo": "https://huggingface.co/HuggingFaceTB/SmolLM2-360M" }, "_bio_source": "groq", "_card_source": "raw:main" }, "HuggingFaceTB/SmolLM2-135M": { "star_name": "SmolLM2-135M", "title": "Stellar SmolLM Beacon", "motto": "Lightweight brilliance, boundless insight", "bio": "SmolLM2-135M, a 134.52 M‑parameter model, excels at instruction following, knowledge, and reasoning while remaining lightweight for on‑device use. Trained on 2 trillion tokens from FineWeb‑Edu, DCLM, The Stack, and curated datasets, it was fine‑tuned with SFT and DPO using UltraFeedback. The model is released under the Apache‑2.0 license.", "author_bio": "Created by the HuggingFace community, SmolLM2 is part of the open‑source AI research ecosystem.", "specialty": "Counterfactual", "spectral_class": "G", "constellation": "The HuggingFaceTB Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffe9a8", "links": { "repo": "https://huggingface.co/HuggingFaceTB/SmolLM2-135M" }, "_bio_source": "groq", "_card_source": "raw:main" }, "facebook/MobileLLM-R1-140M-base": { "star_name": "MobileLLM-R1-140M-base", "title": "Stellar Beacon of Reason", "motto": "Illuminating Precision in Every Query", "bio": "MobileLLM‑R1‑140M‑base is a 140‑million‑parameter efficient reasoning model from Facebook. It is a supervised fine‑tuned base model trained for mathematical, programming, and scientific tasks, and is part of the MobileLLM‑R1 family. The model is released under an other license and uses the Transformers library.", "author_bio": "Developed by Facebook AI Research.", "specialty": "Counterfactual", "spectral_class": "G", "constellation": "The facebook Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffe9a8", "links": { "repo": "https://huggingface.co/facebook/MobileLLM-R1-140M-base" }, "_bio_source": "groq", "_card_source": "hub" }, "AxiomicLabs/GPT-X2.5-135M": { "star_name": "GPT-X2.5-135M", "title": "Axiom Star GPT-X2.5", "motto": "Precision in 135M", "bio": "GPT-X2.5-135M is a 135 million-parameter transformer built on the TX-3 architecture with 30 layers and a custom 32K tokenizer. It employs XGQA attention and was trained from scratch on a multi-source curriculum, achieving near state-of-the-art results on language and mathematical reasoning benchmarks. With 75 billion training tokens, it reaches a 25.17 Intelligence Index—within 1.96 points of SmolLM2-135M—while using about 27× fewer tokens.", "author_bio": "Axiomic Labs, a research organization focused on advancing small language models, developed GPT-X2.5-135M.", "specialty": "State Tracking", "spectral_class": "G", "constellation": "The AxiomicLabs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffe9a8", "links": { "repo": "https://huggingface.co/AxiomicLabs/GPT-X2.5-135M" }, "_bio_source": "groq", "_card_source": "raw:main" }, "BananaMind/BananaMind-2-Pro": { "star_name": "BananaMind-2-Pro", "title": "BananaMind Stellar Model", "motto": "Decoding the Future", "bio": "BananaMind-2-Pro is a 138,971,520‑parameter decoder‑only model trained from scratch. It processed 99,999,449,088 tokens over 184,954 optimizer steps and supports a 3,072‑token context window with a 32,768‑token byte‑level BPE tokenizer. The architecture features grouped‑query attention, QK normalization, RoPE, SwiGLU, RMSNorm, tied embeddings, and a KV cache.", "author_bio": "Developed by BananaMind, a research lab focused on large language models.", "specialty": "Counterfactual", "spectral_class": "G", "constellation": "The BananaMind Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffe9a8", "links": { "repo": "https://huggingface.co/BananaMind/BananaMind-2-Pro" }, "_bio_source": "groq", "_card_source": "raw:main" }, "SupraLabs/Supra2-100M-Base": { "star_name": "Supra2-100M-Base", "title": "Supra 100M Stellar Model", "motto": "Uncharted horizons in language decoding", "bio": "Supra2-100M Base is a 100.68‑million‑parameter decoder‑only model built on the Qwen3 architecture with a 32,768‑token custom tokenizer. Trained from scratch on 30 billion English web tokens, it remains a pure base model, unaligned or instruction‑tuned. The model is released under the Apache‑2.0 license via the transformers library.", "author_bio": "SupraLabs, a research group focused on scalable language models, developed Supra2-100M Base.", "specialty": "Counterfactual", "spectral_class": "G", "constellation": "The SupraLabs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffe9a8", "links": { "repo": "https://huggingface.co/SupraLabs/Supra2-100M-Base" }, "_bio_source": "groq", "_card_source": "raw:main" }, "AxiomicLabs/GPT-X2-125M": { "star_name": "GPT-X2-125M", "title": "Stellar Beacon of State Tracking", "motto": "Precision in Every Pulse", "bio": "GPT‑X2‑125M, a 125‑million‑parameter model, excels in state tracking with a CI‑Index of 35.13. Built on the T‑X2 architecture, it matches SmolLM2‑135M on aggregate while using 27× fewer tokens. Trained from scratch on a 75‑billion‑token curriculum, it achieves near state‑of‑the‑art performance.", "author_bio": "Authored by Axiomic Labs, a pioneer in efficient language model research.", "specialty": "State Tracking", "spectral_class": "G", "constellation": "The AxiomicLabs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffe9a8", "links": { "repo": "https://huggingface.co/AxiomicLabs/GPT-X2-125M" }, "_bio_source": "groq", "_card_source": "raw:main" }, "SurjoLabs/Flare": { "star_name": "Flare", "title": "Flare Star of State Tracking", "motto": "Recurrent brilliance, boundless insight", "bio": "Flare, a 130‑million‑parameter model by SurjoLabs, showcases a custom XSA recurrent architecture that rivals transformers in reasoning. Trained on 26.21 billion tokens with a 32,768‑vocabulary tokenizer, it excels in state tracking and mathematical inference.", "author_bio": "SurjoLabs, a pioneer in efficient language model research, authored Flare.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The SurjoLabs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/SurjoLabs/Flare" }, "_bio_source": "groq", "_card_source": "raw:main" }, "fromziro/Zero-v0.1-150M": { "star_name": "Zero-v0.1-150M", "title": "Zero Star of Counterfactuals", "motto": "Illuminating possibilities beyond known horizons", "bio": "Zero-v0.1-150M is the first beta checkpoint of the Zero model family, featuring 34 layers and 151.6 million parameters. Trained on 69.6 billion tokens, it excels in counterfactual tasks but currently trails larger baselines like SmolLM2 and MobileLLM. The model uses a 2048-token context window and a 32‑dimensional head.", "author_bio": "Developed by the fromziro research team under the Apache-2.0 license.", "specialty": "Counterfactual", "spectral_class": "K", "constellation": "The fromziro Nebula", "hardware": "Tesla T4", "tokens_seen": 69601930240, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/fromziro/Zero-v0.1-150M" }, "_bio_source": "groq", "_card_source": "raw:main" }, "BananaMind/BananaMind-2.1-Unified": { "star_name": "BananaMind-2.1-Unified", "title": "BananaMind-2.1-Unified: Relay Star", "motto": "Three towers, one shared destiny", "bio": "BananaMind-2.1-Unified is a 35‑million‑parameter, three‑tower decoder‑only causal language model. It shares a 384‑dimensional embedding across towers A, B (relay) and C, with A and C providing output heads whose token probabilities are mixed. Trained from scratch on a 38‑billion‑token flat mix, it uses RoPE, grouped‑query attention, and a SwiGLU MLP.", "author_bio": "Developed by BananaMind, a research collective advancing open‑source language models.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The BananaMind Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/BananaMind/BananaMind-2.1-Unified" }, "_bio_source": "groq", "_card_source": "raw:main" }, "SurjoLabs/Blaze": { "star_name": "Blaze", "title": "Blaze Star of SurjoLabs", "motto": "Igniting Language, Illuminating Minds", "bio": "Blaze is a 48.3‑million‑parameter causal language model from SurjoLabs that tops the Open SLM Leaderboard’s sub‑50 M category with a 15.45 Intelligence Index score. Its XSA attention and recurrent layer sharing give it an effective depth of 26 layers while only storing 14 physical layers, trained on ~20.97 B tokens with a WSD scheduler.", "author_bio": "SurjoLabs is a research organization dedicated to advancing small language models.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The SurjoLabs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/SurjoLabs/Blaze" }, "_bio_source": "groq", "_card_source": "raw:main" }, "GODELEV/Rose-Medium": { "star_name": "Rose-Medium", "title": "Rose Nebula Luminary", "motto": "Tracking the state of the cosmos", "bio": "Rose-Medium is a 97.82‑million‑parameter model built on the Rose X1 architecture and trained on approximately 80 billion tokens. It features a native 2048‑token context window and is released under the Apache‑2.0 license for use with the transformers library. The model excels in state‑tracking tasks, earning a CI‑Index of 34.0/100.", "author_bio": "Developed by the GODELEV research team.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The GODELEV Nebula", "hardware": "Tesla T4", "tokens_seen": 80063705088, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/GODELEV/Rose-Medium" }, "_bio_source": "groq", "_card_source": "raw:main" }, "bench-labs/cagliostro-v1": { "star_name": "cagliostro-v1", "title": "Cagliostro Stellar Decoding", "motto": "Decoder-only 36B tokens 157M stars", "bio": "It is a 157M parameter decoder-only language model trained from scratch on 36B tokens of web text. It has not been instruction tuned and will not follow instructions or hold a conversation. It achieved an Intelligence Index of 17.85 on the Open SLM Leaderboard.", "author_bio": "Bench Labs develops open-source language models under an MIT license.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The bench-labs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/bench-labs/cagliostro-v1" }, "_bio_source": "groq", "_card_source": "raw:main" }, "GODELEV/Rose-Pro": { "star_name": "Rose-Pro", "title": "Rose Pro: State Tracker Star", "motto": "Tracking the cosmos, one state", "bio": "Rose Pro is a 151.27-million-parameter language model in the Rose X1 family. It uses 24 transformer layers with 640-dimensional hidden states and was trained on roughly 80 billion tokens. The model excels in state-tracking tasks, earning a CI-Index of 33.78/100.", "author_bio": "Developed by GODELEV, a community of open-source AI researchers.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The GODELEV Nebula", "hardware": "Tesla T4", "tokens_seen": 79967283200, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/GODELEV/Rose-Pro" }, "_bio_source": "groq", "_card_source": "raw:main" }, "fromziro/Er-Large-30M": { "star_name": "Er-Large-30M", "title": "Er's Radiant State Tracker", "motto": "Guiding the cosmos with state-of-the-art text", "bio": "Er-Large is a 32‑million‑parameter small language model trained on 34.8 B tokens from a nine‑source dataset. It achieved a CI‑Index of 33.38/100 and tops the State Tracking category. Built on the Qwen3.5 architecture, it was trained for 116 hours with PyTorch and transformers.", "author_bio": "The model was developed by Paul Courneya and Jonathon LY under the Er SLM family, released under the Apache‑2.0 license.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The fromziro Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/fromziro/Er-Large-30M" }, "_bio_source": "groq", "_card_source": "raw:main" }, "openai-community/gpt2": { "star_name": "gpt2", "title": "GPT-2 Stellar Model", "motto": "Predicting the future, one word at a time", "bio": "GPT-2 is a transformer model pretrained on a large corpus of English text using a causal language modeling objective. It was introduced in the paper 'Language Models are Unsupervised Multitask Learners' and released by OpenAI. The model contains 124.44 million parameters and is licensed under MIT.", "author_bio": "This entry was authored by the Stargazer of the SLM-CI Observatory, a leaderboard of small language models styled as a stellar chart.", "specialty": "Counterfactual", "spectral_class": "K", "constellation": "The openai-community Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/openai-community/gpt2" }, "_bio_source": "groq", "_card_source": "raw:main" }, "BananaMind/BananaMind-2-Medium": { "star_name": "BananaMind-2-Medium", "title": "BananaMind Stellar Decoder", "motto": "Decoding the cosmos with byte‑level precision", "bio": "BananaMind-2-Medium is a decoder‑only Transformer with 49,559,552 parameters and 12 layers. It employs a 12,288-token byte‑level BPE tokenizer, 3,072-token context window, grouped‑query attention with QK normalization, RoPE (theta 100,000) and RMSNorm (ε 1e‑6). Trained on a 50 B‑token curriculum, it supports KV cache generation.", "author_bio": "BananaMind is an independent research group dedicated to developing open‑source language models.", "specialty": "Counterfactual", "spectral_class": "K", "constellation": "The BananaMind Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/BananaMind/BananaMind-2-Medium" }, "_bio_source": "groq", "_card_source": "raw:main" }, "facebook/opt-125m": { "star_name": "opt-125m", "title": "Optic 125-M Star", "motto": "Illuminating state-tracking horizons", "bio": "The facebook/opt-125m model contains 125.24 million parameters and was released by Meta AI on May 3 2022. It excels in state-tracking tasks and is part of the OPT decoder-only transformer series. The model card, authored by Hugging Face, is licensed under a non-standard license.", "author_bio": "The author is the Hugging Face team, curating open-source AI models.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The facebook Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/facebook/opt-125m" }, "_bio_source": "groq", "_card_source": "raw:main" }, "GODELEV/Rose-Mini": { "star_name": "Rose-Mini", "title": "Rose of the Mini Galaxy", "motto": "Blooming with State Tracking Brilliance", "bio": "Rose-Mini, a 49.44M-parameter model, leads the State Tracking category with a CI-Index of 31.18/100 after training on 12.2B tokens. Developed under an Apache-2.0 license, it is the first public release built on the custom Rose X1 architecture.", "author_bio": "The model was created by GODELEV, a developer focused on innovative language model architectures.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The GODELEV Nebula", "hardware": "Tesla T4", "tokens_seen": 12206861568, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/GODELEV/Rose-Mini" }, "_bio_source": "groq", "_card_source": "raw:main" }, "veyra-ai/Veyra2-Mango-15M-Base": { "star_name": "Veyra2-Mango-15M-Base", "title": "Veyra Mango 15M Star", "motto": "Precision in Every Token", "bio": "Veyra2-Mango-15M-Base is a 15.7-million-parameter Llama-like causal language model with 8 transformer layers and a 384-dimensional hidden size. It was trained from scratch on roughly 30 billion tokens and is intended for research, benchmarking, and small-model experimentation.", "author_bio": "This model was developed by Veyra AI, an organization dedicated to advancing open-source language models.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The veyra-ai Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/veyra-ai/Veyra2-Mango-15M-Base" }, "_bio_source": "groq", "_card_source": "raw:main" }, "SupraLabs/Supra2-Medium-Base": { "star_name": "Supra2-Medium-Base", "title": "Star of Data Efficiency", "motto": "Efficiency in the Light of Language", "bio": "Supra2-Medium Base is a 25-million-parameter decoder-only model built on the Qwen3 architecture. Trained from scratch on 20 billion English web tokens, it achieves ~800 tokens per parameter, surpassing typical pretraining ratios. Its lightweight design makes it ideal for data-efficient scaling research and ultra-lightweight deployments.", "author_bio": "SupraLabs, a research organization dedicated to advancing efficient language models, created Supra2-Medium Base.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The SupraLabs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/SupraLabs/Supra2-Medium-Base" }, "_bio_source": "groq", "_card_source": "raw:main" }, "veyra-ai/Veyra2-Apricot-50M-Base": { "star_name": "Veyra2-Apricot-50M-Base", "title": "Veyra Apricot Nebula", "motto": "Harnessing 20B tokens, 49M parameters.", "bio": "Veyra2‑Apricot‑50M‑Base is a 49.3‑million‑parameter Llama‑like causal language model. Trained from scratch on ~20 B tokens with bfloat16 precision and the Muon optimizer, it features 16 layers, 512‑dim hidden states, and 8 attention heads. The model is intended for research, benchmarking, and small‑model experimentation.", "author_bio": "Developed by the Veyra AI research team at Veyra.ai.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The veyra-ai Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/veyra-ai/Veyra2-Apricot-50M-Base" }, "_bio_source": "groq", "_card_source": "raw:main" }, "SupraLabs/Supra-50M-Base": { "star_name": "Supra-50M-Base", "title": "Supra-50M Stellar Beacon", "motto": "Compact power, boundless insight", "bio": "Supra-50M is a 51.79‑million‑parameter causal language model built by SupraLabs. Trained from scratch on 20 billion educational web tokens with a Llama‑style architecture, it scores 76.3 % on BLiMP, 77.2 % on SciQ, 52.2 % on ARC‑Easy, 62.2 % on PIQA and 31.8 % on HellaSwag.", "author_bio": "SupraLabs is a research organization pioneering efficient language models.", "specialty": "Counterfactual", "spectral_class": "K", "constellation": "The SupraLabs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/SupraLabs/Supra-50M-Base" }, "_bio_source": "groq", "_card_source": "raw:main" }, "BananaMind/BananaMind-1.5-Base": { "star_name": "BananaMind-1.5-Base", "title": "BananaMind Star of the Llama Nebula", "motto": "Harnessing 27B tokens, blazing 75M stars", "bio": "BananaMind-1.5-Base is a 75,054,720‑parameter Llama‑style decoder‑only transformer trained from scratch on roughly 27 billion tokens. It offers 12 layers with a 640‑dim hidden size, 10 attention heads, a 4096‑token context window, and was trained on an RTX Pro 6000 using BF16 precision at a cost of $103.31.", "author_bio": "BananaMind is an independent AI research lab focused on building accessible, high‑performance language models.", "specialty": "State Tracking", "spectral_class": "K", "constellation": "The BananaMind Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ffb46b", "links": { "repo": "https://huggingface.co/BananaMind/BananaMind-1.5-Base" }, "_bio_source": "groq", "_card_source": "raw:main" }, "DALabCommunity/Haidass1.5-143M": { "star_name": "Haidass1.5-143M", "title": "Stellar Laureate of Ascend", "motto": "Beyond Language, Beyond Stars", "bio": "Haidass1.5-143M is a 143 M‑parameter bilingual model that achieved 4th place on the OpenSLM Leaderboard and 2nd on Tiny‑ML-Leaderboard. Trained entirely on the Ascend ecosystem with roughly 400 B tokens, it excels in syllogistic reasoning and multilingual tasks.", "author_bio": "The model was developed by the DALabCommunity, a collaborative research group focused on advancing small language models.", "specialty": "Syllogism", "spectral_class": "M", "constellation": "The DALabCommunity Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ff7a59", "links": { "repo": "https://huggingface.co/DALabCommunity/Haidass1.5-143M" }, "_bio_source": "groq", "_card_source": "raw:main" }, "Dummy/Dummy": { "star_name": "Dummy", "title": "Stellar Dummy of Algorithmic Brilliance", "motto": "Algorithmic Light Unseen Depth", "bio": "Dummy/Dummy is a 44.1‑million‑parameter model that shines brightest in the Algorithmic category. Its CI‑Index stands at 23.4 out of 100, reflecting moderate consistency. Training token count remains undisclosed.", "author_bio": "Authored by the SLM‑CI Observatory team, chronicling the cosmos of language models.", "specialty": "Algorithmic", "spectral_class": "M", "constellation": "The Dummy Nebula", "hardware": "", "tokens_seen": null, "emblem_color": "#ff7a59", "links": { "repo": "https://huggingface.co/Dummy/Dummy" }, "_bio_source": "groq", "_card_source": null }, "specklabs/Speck1-140M": { "star_name": "Speck1-140M", "title": "Speck of the Quantum Sky", "motto": "Attention and convolution, united", "bio": "Speck1-140M is a 140.7M‑parameter English base language model that interleaves global grouped‑query attention with gated causal convolution. Pretrained from scratch on 5 B tokens, it delivers a validation perplexity of 10.65 and runs at 247 tok/s on an RTX 3090. The model is released under the MIT license in BF16 Safetensors format.", "author_bio": "Developed by Speck Labs, a research group dedicated to efficient language models.", "specialty": "Spatial", "spectral_class": "M", "constellation": "The specklabs Nebula", "hardware": "Tesla T4", "tokens_seen": null, "emblem_color": "#ff7a59", "links": { "repo": "https://huggingface.co/specklabs/Speck1-140M" }, "_bio_source": "groq", "_card_source": "raw:main" } }