Supernova Transformer V2

Supernova Transformer V2 is a character-level decoder-only Transformer trained on the Supernova conversational dataset.

Architecture

  • Vocabulary: 131
  • Hidden size: 256
  • Transformer layers: 4
  • Attention heads: 8
  • Feed-forward dimension: 1024
  • Maximum sequence length: 512
  • Trainable parameters: 3,357,696

The token embedding and language-model head are separate parameter tensors. Their values are identical in the original checkpoint, but they are not parameter-tied.

Training

Training examples: 2601 Validation examples: 289

Best validation loss:

0.142253234871896

Validation perplexity:

1.152869

Special Tokens

Token ID
<PAD> 0
<BOS> 1
<USER> 2
<ASSISTANT> 3
<EOS> 4

There is no <UNK> token.

Exact Export Verification

The exported model.safetensors was verified tensor-by-tensor against the original PyTorch checkpoint.

The reconstructed architecture was loaded using a strict state-dict load with zero missing or unexpected keys.

Attention parameters preserve the original checkpoint names:

  • attn.qkv.weight
  • attn.qkv.bias
  • attn.out.weight
  • attn.out.bias

Feed-forward parameters preserve:

  • ff.0
  • ff.1
  • ff.2

The exported model successfully completes a CPU forward pass with finite logits.

Model Scope

This model was trained primarily on Supernova-specific conversational behavior, responses, and project-related data. It should not be interpreted as a broad general-world-knowledge language model.

# ================================================================================
# SUPERNOVA TRANSFORMER V2
# HUGGING FACE REMOTE WEIGHTS — FRAMEWORK-INDEPENDENT TEST
# ================================================================================

!pip -q install -U huggingface_hub safetensors

import os
import sys
import json
import math
import torch
import torch.nn as nn
import torch.nn.functional as F

from pathlib import Path
from huggingface_hub import hf_hub_download
from safetensors.torch import load_file


REPO_ID = "Supernova11c/Supernova-Transformer-V2"

print("=" * 80)
print("SUPERNOVA TRANSFORMER V2")
print("REMOTE HUGGING FACE WEIGHTS TEST")
print("=" * 80)


# ================================================================================
# 1. DOWNLOAD PUBLISHED FILES
# ================================================================================

print("\n" + "=" * 80)
print("DOWNLOADING FROM HUGGING FACE")
print("=" * 80)

config_path = hf_hub_download(
    repo_id=REPO_ID,
    filename="config.json",
)

vocab_path = hf_hub_download(
    repo_id=REPO_ID,
    filename="vocab.json",
)

model_path = hf_hub_download(
    repo_id=REPO_ID,
    filename="model.safetensors",
)

print("✓ config.json downloaded")
print("✓ vocab.json downloaded")
print("✓ model.safetensors downloaded")


# ================================================================================
# 2. LOAD CONFIG
# ================================================================================

with open(
    config_path,
    "r",
    encoding="utf-8",
) as f:
    config = json.load(f)

print("\nConfiguration:")
print("  vocab_size :", config["vocab_size"])
print("  d_model    :", config["d_model"])
print("  n_heads    :", config["n_heads"])
print("  n_layers   :", config["n_layers"])
print("  ffn_dim    :", config["ffn_dim"])

assert config["vocab_size"] == 131
assert config["d_model"] == 256
assert config["n_heads"] == 8
assert config["n_layers"] == 4
assert config["ffn_dim"] == 1024
assert config["tie_word_embeddings"] is False

print("✓ Remote config verified.")


# ================================================================================
# 3. LOAD EXACT PUBLISHED SAFETENSORS
# ================================================================================

print("\n" + "=" * 80)
print("LOADING PUBLISHED WEIGHTS")
print("=" * 80)

remote_state = load_file(
    model_path,
    device="cpu",
)

print("Remote tensors:", len(remote_state))

assert len(remote_state) == 53

remote_elements = sum(
    tensor.numel()
    for tensor in remote_state.values()
)

assert remote_elements == 3_357_696

print("✓ 53 tensors loaded.")
print("✓ 3,357,696 parameters loaded.")


# ================================================================================
# 4. EXACT TENSOR SHAPES
# ================================================================================

expected_shapes = {
    "token_embedding.weight": (131, 256),
    "position_embedding.weight": (512, 256),

    "ln_final.weight": (256,),
    "ln_final.bias": (256,),

    "lm_head.weight": (131, 256),
}

for key, shape in expected_shapes.items():

    assert key in remote_state
    assert tuple(remote_state[key].shape) == shape

for i in range(4):

    prefix = f"blocks.{i}"

    assert tuple(
        remote_state[f"{prefix}.ln1.weight"].shape
    ) == (256,)

    assert tuple(
        remote_state[f"{prefix}.attn.qkv.weight"].shape
    ) == (768, 256)

    assert tuple(
        remote_state[f"{prefix}.attn.qkv.bias"].shape
    ) == (768,)

    assert tuple(
        remote_state[f"{prefix}.attn.out.weight"].shape
    ) == (256, 256)

    assert tuple(
        remote_state[f"{prefix}.attn.out.bias"].shape
    ) == (256,)

    assert tuple(
        remote_state[f"{prefix}.ff.0.weight"].shape
    ) == (1024, 256)

    assert tuple(
        remote_state[f"{prefix}.ff.2.weight"].shape
    ) == (256, 1024)

print("✓ Architecture tensor shapes verified.")


# ================================================================================
# 5. EXACT TOKENIZER
# ================================================================================

with open(
    vocab_path,
    "r",
    encoding="utf-8",
) as f:
    vocab = json.load(f)

vocab = {
    str(k): int(v)
    for k, v in vocab.items()
}

assert len(vocab) == 131

itos = {
    idx: token
    for token, idx in vocab.items()
}

assert vocab["<PAD>"] == 0
assert vocab["<BOS>"] == 1
assert vocab["<USER>"] == 2
assert vocab["<ASSISTANT>"] == 3
assert vocab["<EOS>"] == 4

print("✓ Published vocabulary verified.")


def encode(text):
    ids = []

    for char in text:
        if char not in vocab:
            raise ValueError(
                f"Character not in V2 vocabulary: {char!r}"
            )

        ids.append(vocab[char])

    return ids


def decode(ids):
    return "".join(
        itos[int(i)]
        for i in ids
    )


# ================================================================================
# 6. RECREATE EXACT MODEL ARCHITECTURE
# ================================================================================

class SupernovaSelfAttention(nn.Module):

    def __init__(
        self,
        d_model,
        n_heads,
        dropout=0.0,
    ):
        super().__init__()

        assert d_model % n_heads == 0

        self.d_model = d_model
        self.n_heads = n_heads
        self.head_dim = d_model // n_heads

        self.qkv = nn.Linear(
            d_model,
            3 * d_model,
            bias=True,
        )

        self.out = nn.Linear(
            d_model,
            d_model,
            bias=True,
        )

        self.dropout = nn.Dropout(dropout)


    def forward(
        self,
        x,
        causal_mask=None,
    ):

        B, T, C = x.shape

        qkv = self.qkv(x)

        q, k, v = qkv.chunk(
            3,
            dim=-1,
        )

        q = q.view(
            B,
            T,
            self.n_heads,
            self.head_dim,
        ).transpose(1, 2)

        k = k.view(
            B,
            T,
            self.n_heads,
            self.head_dim,
        ).transpose(1, 2)

        v = v.view(
            B,
            T,
            self.n_heads,
            self.head_dim,
        ).transpose(1, 2)

        scores = torch.matmul(
            q,
            k.transpose(-2, -1),
        ) / math.sqrt(
            self.head_dim
        )

        if causal_mask is not None:

            scores = scores.masked_fill(
                causal_mask.view(
                    1,
                    1,
                    T,
                    T,
                ),
                float("-inf"),
            )

        weights = F.softmax(
            scores,
            dim=-1,
        )

        y = torch.matmul(
            weights,
            v,
        )

        y = y.transpose(
            1,
            2,
        ).contiguous().view(
            B,
            T,
            C,
        )

        return self.out(y)


class SupernovaBlock(nn.Module):

    def __init__(
        self,
        d_model,
        n_heads,
        ffn_dim,
    ):
        super().__init__()

        self.ln1 = nn.LayerNorm(
            d_model
        )

        self.attn = SupernovaSelfAttention(
            d_model,
            n_heads,
        )

        self.ln2 = nn.LayerNorm(
            d_model
        )

        self.ff = nn.Sequential()

        self.ff.add_module(
            "0",
            nn.Linear(
                d_model,
                ffn_dim,
            ),
        )

        self.ff.add_module(
            "1",
            nn.GELU(),
        )

        self.ff.add_module(
            "2",
            nn.Linear(
                ffn_dim,
                d_model,
            ),
        )


    def forward(
        self,
        x,
        causal_mask,
    ):

        x = x + self.attn(
            self.ln1(x),
            causal_mask,
        )

        x = x + self.ff(
            self.ln2(x)
        )

        return x


class SupernovaV2(nn.Module):

    def __init__(
        self,
        config,
    ):
        super().__init__()

        self.vocab_size = config["vocab_size"]
        self.d_model = config["d_model"]
        self.n_heads = config["n_heads"]
        self.n_layers = config["n_layers"]
        self.ffn_dim = config["ffn_dim"]
        self.max_seq_len = config["max_seq_len"]

        self.token_embedding = nn.Embedding(
            self.vocab_size,
            self.d_model,
        )

        self.position_embedding = nn.Embedding(
            self.max_seq_len,
            self.d_model,
        )

        self.blocks = nn.ModuleList([
            SupernovaBlock(
                self.d_model,
                self.n_heads,
                self.ffn_dim,
            )
            for _ in range(self.n_layers)
        ])

        self.ln_final = nn.LayerNorm(
            self.d_model
        )

        self.lm_head = nn.Linear(
            self.d_model,
            self.vocab_size,
            bias=False,
        )


    def forward(self, input_ids):

        B, T = input_ids.shape

        assert T <= self.max_seq_len

        positions = torch.arange(
            T,
            device=input_ids.device,
        )

        x = (
            self.token_embedding(input_ids)
            +
            self.position_embedding(
                positions
            )[None, :, :]
        )

        causal_mask = torch.triu(
            torch.ones(
                T,
                T,
                dtype=torch.bool,
                device=input_ids.device,
            ),
            diagonal=1,
        )

        for block in self.blocks:

            x = block(
                x,
                causal_mask,
            )

        x = self.ln_final(x)

        logits = self.lm_head(x)

        return logits


# ================================================================================
# 7. LOAD REMOTE WEIGHTS STRICTLY
# ================================================================================

print("\n" + "=" * 80)
print("STRICT MODEL LOAD")
print("=" * 80)

model = SupernovaV2(config)

missing, unexpected = model.load_state_dict(
    remote_state,
    strict=False,
)

assert missing == []
assert unexpected == []

model.eval()

print("✓ Published weights loaded into exact architecture.")
print("✓ Missing keys   :", missing)
print("✓ Unexpected keys:", unexpected)


# ================================================================================
# 8. FORWARD PASS
# ================================================================================

print("\n" + "=" * 80)
print("FORWARD PASS")
print("=" * 80)

test_text = "Supernova के हो?"

input_ids = torch.tensor(
    [[
        vocab["<BOS>"],
        vocab["<USER>"],
        *encode(test_text),
        vocab["<ASSISTANT>"],
    ]],
    dtype=torch.long,
)

with torch.no_grad():

    logits = model(
        input_ids
    )

print("Input shape :", tuple(input_ids.shape))
print("Logits shape:", tuple(logits.shape))

assert logits.shape == (
    1,
    input_ids.shape[1],
    131,
)

assert torch.isfinite(logits).all()

print("✓ Forward pass successful.")
print("✓ All logits finite.")


# ================================================================================
# 9. GREEDY GENERATION
# ================================================================================

print("\n" + "=" * 80)
print("REAL GENERATION")
print("=" * 80)


def generate(
    prompt,
    max_new_tokens=180,
):

    ids = [
        vocab["<BOS>"],
        vocab["<USER>"],
    ]

    ids += encode(prompt)

    ids += [
        vocab["<ASSISTANT>"]
    ]

    generated = torch.tensor(
        [ids],
        dtype=torch.long,
    )

    for _ in range(max_new_tokens):

        context = generated[
            :, -512:
        ]

        with torch.no_grad():

            logits = model(
                context
            )

        next_logits = logits[
            :, -1, :
        ]

        # Prevent generation of structural tokens.
        for token in [
            vocab["<PAD>"],
            vocab["<BOS>"],
            vocab["<USER>"],
            vocab["<ASSISTANT>"],
        ]:

            next_logits[:, token] = -float(
                "inf"
            )

        next_id = torch.argmax(
            next_logits,
            dim=-1,
        ).item()

        generated = torch.cat(
            [
                generated,
                torch.tensor(
                    [[next_id]],
                    dtype=torch.long,
                ),
            ],
            dim=1,
        )

        if next_id == vocab["<EOS>"]:
            break

    answer_ids = generated[
        0,
        len(ids):,
    ].tolist()

    return decode(answer_ids)


tests = [
    "Supernova के हो?",
    "What is Supernova?",
    "Supernova को उद्देश्य के हो?",
    "What can you do?",
    "What is Python?",
    "मलाई Python बारे बताउनुहोस्।",
    "यदि तिमीलाई उत्तर थाहा छैन भने के गर्छौ?",
]

for i, prompt in enumerate(
    tests,
    1,
):

    print("\n" + "-" * 80)
    print(f"TEST {i}")
    print("USER:", prompt)

    answer = generate(
        prompt,
        max_new_tokens=180,
    )

    print("ASSISTANT:", answer)

    assert len(answer) > 0

    print("✓ Generated successfully.")


# ================================================================================
# 10. FINAL
# ================================================================================

print("\n" + "=" * 80)
print("REMOTE HUGGING FACE TEST COMPLETE")
print("=" * 80)

print("✓ Remote config       : PASS")
print("✓ Remote vocabulary   : PASS")
print("✓ Remote safetensors  : PASS")
print("✓ 53 tensors          : PASS")
print("✓ 3,357,696 params    : PASS")
print("✓ Strict architecture : PASS")
print("✓ Forward pass        : PASS")
print("✓ Real generation     : TESTED")
print("=" * 80)
Downloads last month
170
Safetensors
Model size
3.36M params
Tensor type
F32
·
Inference Providers NEW
This model isn't deployed by any Inference Provider. 🙋 Ask for provider support