Download nexora/tokenizer.py from devildasdf/NEXORA: direct link, hf CLI and curl.
- Browser
- Download file 1.11 kB
-
https://huggingface.co/devildasdf/NEXORA/resolve/main/nexora/tokenizer.py
- Command line
-
hf download hf://devildasdf/NEXORA/nexora/tokenizer.py
-
curl -L -o tokenizer.py https://huggingface.co/devildasdf/NEXORA/resolve/main/nexora/tokenizer.py
1.11 kB
| """Lossless byte baseline; not a claim of competitive token compression.""" | |
| from dataclasses import dataclass | |
| class ByteTokenizer: | |
| pad_id: int = 256 | |
| bos_id: int = 257 | |
| eos_id: int = 258 | |
| vocab_size: int = 259 | |
| def encode(self, text: str, *, special: bool = False) -> list[int]: | |
| ids = list(text.encode("utf-8")) | |
| return [self.bos_id, *ids, self.eos_id] if special else ids | |
| def decode(self, ids: list[int]) -> str: | |
| if any(not isinstance(i, int) or not 0 <= i < self.vocab_size for i in ids): | |
| raise ValueError("Token outside vocabulary") | |
| return bytes(i for i in ids if i < 256).decode("utf-8", errors="replace") | |
| def benchmark(self, samples: dict[str, str]) -> dict: | |
| return {name: {"characters": len(text), "bytes": len(text.encode()), | |
| "tokens": len(self.encode(text)), | |
| "characters_per_token": len(text) / max(1, len(self.encode(text))), | |
| "roundtrip": self.decode(self.encode(text)) == text} | |
| for name, text in samples.items()} | |