NEXORA / nexora /tokenizer.py
devildasdf's picture
Release validated NEXORA research prototype, tiny weights and evidence
12496fc verified
Raw History Blame Contribute Delete
1.11 kB
"""Lossless byte baseline; not a claim of competitive token compression."""
from dataclasses import dataclass
@dataclass(frozen=True)
class ByteTokenizer:
pad_id: int = 256
bos_id: int = 257
eos_id: int = 258
vocab_size: int = 259
def encode(self, text: str, *, special: bool = False) -> list[int]:
ids = list(text.encode("utf-8"))
return [self.bos_id, *ids, self.eos_id] if special else ids
def decode(self, ids: list[int]) -> str:
if any(not isinstance(i, int) or not 0 <= i < self.vocab_size for i in ids):
raise ValueError("Token outside vocabulary")
return bytes(i for i in ids if i < 256).decode("utf-8", errors="replace")
def benchmark(self, samples: dict[str, str]) -> dict:
return {name: {"characters": len(text), "bytes": len(text.encode()),
"tokens": len(self.encode(text)),
"characters_per_token": len(text) / max(1, len(self.encode(text))),
"roundtrip": self.decode(self.encode(text)) == text}
for name, text in samples.items()}