"""How Claude segments text, next to the tokenizers it is usually compared against."""
import html
import re
from dataclasses import dataclass
from functools import lru_cache
import gradio as gr
import pandas as pd
import spaces
import tiktoken
from ctok import pieces, tokenize
from tokenizers import Tokenizer
@spaces.GPU(duration=1)
def _zero_gpu_marker() -> None:
"""Never called, and nothing here wants a GPU — tokenizing is string work on the CPU.
Free Gradio Spaces run on ZeroGPU, which refuses to start unless it finds at least one
`@spaces.GPU` function at import time; CPU Basic is behind PRO. This satisfies that check.
Delete it if the Space ever moves to CPU hardware.
"""
# ── the tokenizers on offer ──────────────────────────────────────────────────
# v4.7 and v5 share a vocabulary and differ only in the message frame, so on text-only counts
# they are the same tokenizer; o200k_harmony likewise only adds special tokens to o200k_base.
CTOK_VERSIONS = {
"Claude v3": "3.0",
"Claude v5": "5.0",
}
TIKTOKEN_ENCODINGS = {
"tiktoken cl100k_base": "cl100k_base",
"tiktoken o200k_base": "o200k_base",
}
LOCAL_CHOICES = list(CTOK_VERSIONS) + list(TIKTOKEN_ENCODINGS)
LOCAL_DEFAULT = ["Claude v5", "tiktoken o200k_base"]
# Suggestions only — the dropdown takes any repo id that ships a `tokenizer.json`.
HF_SUGGESTIONS = [
"deepseek-ai/DeepSeek-V4-Flash",
"deepseek-ai/DeepSeek-V4-Pro",
"deepseek-ai/DeepSeek-V3.2",
"Qwen/Qwen3-8B",
"openai/gpt-oss-120b",
"mistralai/Ministral-8B-Instruct-2410",
"HuggingFaceTB/SmolLM3-3B",
]
HF_DEFAULT = ["deepseek-ai/DeepSeek-V4-Flash"]
MAX_RENDERED_TOKENS = 2000
MAX_TEXT_CHARS = 200_000 # ctok costs ~1.2ms/KB, and this runs on a shared CPU
PAD = "⟨pad⟩"
# ── one uniform result shape ─────────────────────────────────────────────────
@dataclass
class Tally:
"""What the table shows about one tokenizer: everything but the pieces themselves."""
name: str
count: int
vocab: str
@dataclass
class Segmentation:
"""One tokenizer's reading of one text."""
name: str
pieces: list[str]
vocab: str
@property
def count(self) -> int:
return len(self.pieces)
def tally(self) -> Tally:
return Tally(self.name, self.count, self.vocab)
@lru_cache(maxsize=32)
def _hf_tokenizer(repo_id: str) -> Tokenizer:
return Tokenizer.from_pretrained(repo_id)
@lru_cache(maxsize=8)
def _tiktoken_encoding(name: str) -> tiktoken.Encoding:
return tiktoken.get_encoding(name)
@lru_cache(maxsize=4)
def _ctok_vocab(version: str) -> str:
"""ctok reconstructs the vocabulary rather than reading it, so say what the number is."""
return f"{len(pieces(version)):,} measured"
def _ctok_segment(name: str, version: str, text: str) -> Segmentation:
"""ctok's token list starts with the message frame; strip it so the count is text-only.
The frame is the leading ⟨pad⟩ run, which is not the same as `token_count("")`: on v3 the
empty string tokenizes to seven pads plus a stray ⟨bow⟩, and stripping by count would eat
the first word of every real text.
"""
tokens = tokenize(text, version)
frame = next((i for i, t in enumerate(tokens) if t != PAD), len(tokens))
return Segmentation(name, tokens[frame:], _ctok_vocab(version))
def _tiktoken_segment(name: str, encoding: str, text: str) -> Segmentation:
enc = _tiktoken_encoding(encoding)
ids = enc.encode(text, disallowed_special=())
# A token can end mid-character; `replace` shows that as U+FFFD rather than hiding it.
parts = [enc.decode_single_token_bytes(i).decode("utf-8", errors="replace") for i in ids]
return Segmentation(name, parts, f"{enc.n_vocab:,}")
def _hf_segment(repo_id: str, text: str) -> Segmentation:
tok = _hf_tokenizer(repo_id)
encoded = tok.encode(text, add_special_tokens=False)
# Offsets index the original text, so they read better than `Ġworld`-style pieces. A token
# can cover only part of a character, and then it carries that whole character's offsets —
# printing the slice again would show 🇳🇱 as "N N L L". Mark those the way tiktoken does.
parts, consumed = [], 0
for (start, end), piece in zip(encoded.offsets, encoded.tokens):
if end <= start:
parts.append(piece) # zero-width span: show the tokenizer's own name for it
elif end <= consumed:
parts.append("�") # wholly inside a character an earlier token already printed
else:
parts.append(text[max(start, consumed) : end])
consumed = end
return Segmentation(repo_id, parts, f"{tok.get_vocab_size():,}")
def segment_all(text: str, local: list[str], repos: list[str]) -> tuple[list[Segmentation], dict[str, str]]:
"""Every selected tokenizer's reading, plus {repo id: why it would not load}."""
results, failures = [], {}
for name in local:
if name in CTOK_VERSIONS:
results.append(_ctok_segment(name, CTOK_VERSIONS[name], text))
else:
results.append(_tiktoken_segment(name, TIKTOKEN_ENCODINGS[name], text))
for repo_id in repos:
repo_id = repo_id.strip()
if not repo_id:
continue
try:
results.append(_hf_segment(repo_id, text))
except Exception as exc: # noqa: BLE001 - a mistyped repo id is user input, not a bug
failures[repo_id] = f"{type(exc).__name__}: {exc}"
return results, failures
# ── presentation ─────────────────────────────────────────────────────────────
PALETTE = [
"199 89% 48%",
"24 95% 53%",
"142 71% 45%",
"339 82% 52%",
"262 83% 58%",
"43 96% 56%",
"188 86% 43%",
"0 84% 60%",
]
CSS = (
"""
.tokviz { display: flex; flex-direction: column; gap: 18px; }
.tk-head { display: flex; align-items: baseline; gap: 10px; margin-bottom: 6px; }
.tk-name { font-weight: 600; }
.tk-count { font-size: 0.85em; opacity: 0.7; }
.tk-tokens {
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
font-size: 0.9em; line-height: 2; white-space: pre-wrap;
/* `anywhere` keeps a pathological token from overflowing without chopping ordinary words,
and `plaintext` lets each line pick its own direction so Arabic and Hebrew read correctly. */
overflow-wrap: anywhere; word-break: normal; unicode-bidi: plaintext;
}
.tk-tokens .tok { border-radius: 3px 3px 0 0; padding: 3px 0; }
/* Markers are strong-LTR letters; isolating them stops each one flipping the run around it. */
.tk-tokens .mk {
font-size: 0.75em; font-weight: 700; vertical-align: 0.1em; unicode-bidi: isolate;
color: hsl(24 95% 36%); opacity: 0.85; padding: 0 1px;
}
.dark .tk-tokens .mk { color: hsl(24 95% 66%); }
.tk-more { font-size: 0.85em; opacity: 0.7; margin-top: 6px; }
.tk-fail { font-size: 0.9em; }
.links { display: flex; flex-wrap: wrap; gap: 8px; margin: 2px 0 10px; }
.links a {
border: 1px solid var(--border-color-primary, rgba(128, 128, 128, 0.4));
border-radius: 999px; padding: 5px 14px; font-size: 0.9em; text-decoration: none;
color: inherit;
}
.links a:hover { border-color: var(--color-accent, #f97316); }
.links code { background: none; padding: 0; font-size: 1em; }
.counts { display: flex; flex-direction: column; margin-bottom: 4px; }
.counts .row {
display: grid; grid-template-columns: 1fr auto; gap: 1px 12px; align-items: baseline;
padding: 7px 10px; border-radius: 5px;
/* a bar as wide as this tokenizer's share of the largest count */
background: linear-gradient(to right, hsl(24 95% 53% / 0.16) var(--pct), transparent var(--pct));
}
.counts .who { font-weight: 600; overflow-wrap: anywhere; }
.counts .n { font-weight: 600; font-size: 1.05em; font-variant-numeric: tabular-nums; }
.counts .meta { grid-column: 1 / -1; font-size: 0.82em; opacity: 0.72; }
"""
# One class per hue rather than an inline style per token: the same page, a quarter the bytes,
# resent on every keystroke.
+ "".join(
f".tk-tokens .c{i} {{ background: hsl({hue} / 0.25); box-shadow: inset 0 -2px 0 hsl({hue} / 0.8); }}\n"
for i, hue in enumerate(PALETTE)
)
)
MARKER = re.compile(r"⟨([^⟩]+)⟩")
# The notation from the write-up: word bounds as regex anchors, case as arrows. Small and
# coloured, they stay readable at a glance without crowding out the text they wrap.
MARKER_GLYPH = {"bow": "^", "eow": "$", "shift": "↑", "caps": "⇪"}
def _visible(piece: str) -> str:
"""Escape for HTML, mark the structural markers, and show the invisible characters."""
out = MARKER.sub(lambda m: f"{MARKER_GLYPH.get(m[1], m[1])}", html.escape(piece))
return out.replace("\n", "↵\n").replace("\t", "→\t")
def render_tokens(results: list[Segmentation], failures: dict[str, str]) -> str:
blocks = []
for result in results:
shown = result.pieces[:MAX_RENDERED_TOKENS]
spans = "".join(
f"{_visible(p)}" for i, p in enumerate(shown)
)
overflow = (
f"
showing the first {MAX_RENDERED_TOKENS:,} of {result.count:,} tokens
"
if len(result.pieces) > len(shown)
else ""
)
blocks.append(
f"{html.escape(result.name)}"
f"{result.count:,} tokens
"
f"
{spans}
{overflow}
"
)
for repo_id, why in failures.items():
blocks.append(
f"⚠️ could not load {html.escape(repo_id)} — {html.escape(why)}
"
)
return f"{''.join(blocks)}
"
def build_table(tallies: list[Tally], chars: int, byte_len: int) -> str:
"""The counts as rows that reflow, rather than a six-column table that needs a sideways
scroll on a phone. Each row is a bar as wide as its share of the largest count."""
if not tallies:
return ""
best = min((t.count for t in tallies if t.count), default=1)
worst = max((t.count for t in tallies), default=1) or 1
rows = []
for r in tallies:
ratio = f"{r.count / best:.2f}× vs best" if r.count else "—"
per = f"{chars / r.count:.2f} chars" if r.count else "—"
per_b = f"{byte_len / r.count:.2f} bytes" if r.count else "—"
rows.append(
f""
f"{html.escape(r.name)}"
f"{r.count:,}"
f"{ratio} · {per}/token · {per_b}/token · "
f"{r.vocab} pieces"
f"
"
)
return f"{''.join(rows)}
"
# ── tab 1: free text ─────────────────────────────────────────────────────────
def compare_text(text: str, local: list[str], repos: list[str]) -> tuple[str, str, str]:
"""Tokenize one text with every selected tokenizer.
Args:
text: the text to tokenize.
local: built-in tokenizer names, any of ["Claude v3", "Claude v5",
"tiktoken cl100k_base", "tiktoken o200k_base"].
repos: Hugging Face repo ids whose `tokenizer.json` should also be used, for example
["deepseek-ai/DeepSeek-V4-Flash"].
Returns:
A counts table (tokenizer, tokens, ratio against the fewest, characters and bytes per
token, vocabulary size), a one-line size summary of the input, and the segmentation as
HTML. Counts exclude the fixed frame a one-message API request adds.
"""
if not text:
return "", "", ""
clipped = text[:MAX_TEXT_CHARS]
results, failures = segment_all(clipped, local, repos)
table = build_table([r.tally() for r in results], len(clipped), len(clipped.encode("utf-8")))
summary = (
f"{len(clipped):,} characters · {len(clipped.encode('utf-8')):,} UTF-8 bytes"
f" · {len(clipped.split()):,} whitespace words"
)
if len(text) > MAX_TEXT_CHARS:
summary += f" — measuring the first {MAX_TEXT_CHARS:,} of {len(text):,} characters"
return table, summary, render_tokens(results, failures)
# ── tab 2: datasets ──────────────────────────────────────────────────────────
DATASET_SUGGESTIONS = [
"google/wmt24pp",
"wikimedia/wikipedia",
"HuggingFaceFW/fineweb-2",
"HuggingFaceFW/fineweb-edu",
]
@lru_cache(maxsize=64)
def dataset_configs(dataset_id: str) -> list[str]:
"""Configs as the loader sees them, not as the card happens to declare them."""
from datasets import get_dataset_config_names
return list(get_dataset_config_names(dataset_id.strip()))
def _flatten(row: dict, prefix: str = "") -> dict[str, str]:
"""Every string field, including one level down, keyed as `translation.en`."""
out: dict[str, str] = {}
for key, value in row.items():
if isinstance(value, str):
out[prefix + key] = value
elif isinstance(value, dict):
out.update(_flatten(value, f"{prefix}{key}."))
return out
def _field(row: dict, column: str):
"""Read a possibly-dotted column name out of a row."""
value = row
for part in column.split("."):
value = value[part]
return value
def on_dataset_change(dataset_id: str):
if not dataset_id.strip():
return gr.update(choices=[], value=None)
try:
configs = dataset_configs(dataset_id)
except Exception: # noqa: BLE001 - a half-typed dataset id just leaves the picker empty
return gr.update(choices=[], value=None)
return gr.update(choices=configs, value=configs[0] if configs else None)
def _open_stream(dataset_id: str, config: str, split: str):
from datasets import load_dataset
return load_dataset(dataset_id.strip(), config or None, split=split.strip(), streaming=True)
@lru_cache(maxsize=8)
def _stream_rows(dataset_id: str, config: str, split: str, column: str, limit: int) -> tuple[str, ...]:
"""The first `limit` non-empty values of one column.
Cached, so changing the tokenizer lineup and comparing again does not re-download the shard.
The scan is bounded: a mostly-empty column would otherwise read the whole dataset.
"""
rows: list[str] = []
for scanned, row in enumerate(_open_stream(dataset_id, config, split)):
value = _field(row, column)
if isinstance(value, str) and value.strip():
rows.append(value)
if len(rows) >= limit or scanned >= 20 * limit:
break
return tuple(rows)
PEEK_ROWS = 8
def peek_columns(dataset_id: str, config: str, split: str):
"""Fill the column picker, ranked by mean word count over the first few rows.
Not the first row's longest string: row 0 is often a header or canary whose fields are all
the same length, and not length either — an id like `test-en-news.3585` is long and is not
text. Words per row separates prose from identifiers whatever their length.
"""
try:
stream = _open_stream(dataset_id, config, split)
sample = [_flatten(row) for _, row in zip(range(PEEK_ROWS), stream)]
except Exception as exc: # noqa: BLE001 - surfaced in the UI; the hub raises many types
return gr.update(choices=[], value=None), f"⚠️ {type(exc).__name__}: {exc}"
if not sample:
return gr.update(choices=[], value=None), "⚠️ that split is empty"
columns = {key for row in sample for key in row}
if not columns:
return gr.update(choices=[], value=None), "⚠️ no text column in the first rows"
scored = sorted(
columns,
key=lambda c: sum(len(row.get(c, "").split()) for row in sample) / len(sample),
reverse=True,
)
note = " · ".join(
f"{c} ({sum(len(row.get(c, '').split()) for row in sample) // len(sample)} words)" for c in scored[:6]
)
return gr.update(choices=scored, value=scored[0]), f"columns by mean words/row — {note}"
def compare_dataset(
dataset_id: str, config: str, split: str, column: str, rows: int, local: list[str], repos: list[str]
) -> tuple[str, str, gr.BarPlot | None]:
"""Tokenize the first rows of a dataset column with every selected tokenizer.
Args:
dataset_id: a public dataset id, e.g. "google/wmt24pp".
config: the dataset config, e.g. "en-nl_NL".
split: the split to stream, e.g. "train".
column: the text column to read.
rows: how many non-empty rows to read.
local: built-in tokenizer names, any of ["Claude v3", "Claude v5",
"tiktoken cl100k_base", "tiktoken o200k_base"].
repos: Hugging Face repo ids whose `tokenizer.json` should also be used.
Returns:
A totals table over all rows, a summary of what was read, and a characters-per-token
bar chart.
"""
if not dataset_id.strip() or not column:
return "", "Pick a dataset, config and column first.", None
try:
texts = _stream_rows(dataset_id, config, split, column, int(rows))
except Exception as exc: # noqa: BLE001 - surfaced in the UI; the hub raises many types
return "", f"⚠️ {type(exc).__name__}: {exc}", None
if not texts:
return "", "⚠️ no non-empty rows in that column", None
chars = sum(len(t) for t in texts)
byte_len = sum(len(t.encode("utf-8")) for t in texts)
# Only the first row's failures matter: a repo that will not load fails on every row.
_, failures = segment_all(texts[0], [], repos)
working = [r for r in repos if r.strip() not in failures]
totals: dict[str, Tally] = {}
for text in texts:
results, _ = segment_all(text, local, working)
for r in results:
if r.name in totals:
totals[r.name].count += r.count
else:
totals[r.name] = r.tally()
ordered = list(totals.values())
table = build_table(ordered, chars, byte_len)
note = f"{len(texts):,} rows · {chars:,} characters · {byte_len:,} UTF-8 bytes" + "".join(
f"\n\n⚠️ could not load **{repo}** — {why}" for repo, why in failures.items()
)
ratios = [round(chars / r.count, 3) if r.count else 0.0 for r in ordered]
plot = pd.DataFrame({"tokenizer": [r.name for r in ordered], "chars / token": ratios})
# Bars must start at zero, or a 1.6x difference looks like a 20x one.
return (
table,
note,
gr.BarPlot(value=plot, x="tokenizer", y="chars / token", y_lim=[0, max(ratios) * 1.15], visible=True),
)
# ── UI ───────────────────────────────────────────────────────────────────────
REPO_URL = "https://github.com/sanderland/ctok"
PYPI_URL = "https://pypi.org/project/ctok/"
POST_URL = "https://tokencontributions.substack.com/p/on-the-biology-of-claudes-tokenizer"
INTRO = """
# Claude's tokenizer, side by side
Paste text or point at a dataset, and see what each tokenizer charges for it. Claude counts come
from **ctok**, a 99.9%+ accurate offline reconstruction — text only, no per-message API frame.
"""
LINKS = f"""
"""
# The things people actually pay for: other languages, code, JSON payloads, numbers, and the
# Unicode that quietly falls back to bytes.
EXAMPLES = [
# The first five articles of the Universal Declaration, one sentence per language,
# verbatim from the official translations. Same rights, very different token counts.
(
# English
"1. All human beings are born free and equal in dignity and rights.\n"
# French
"Ils sont doués de raison et de conscience et doivent agir les uns envers les autres dans "
"un esprit de fraternité.\n"
# Norwegian
"2. Enhver har krav på alle de rettigheter og friheter som er nevnt i denne erklæring, "
"uten forskjell av noen art, f. eks. på grunn av rase, farge, kjønn, språk, religion, "
"politisk eller annen oppfatning, nasjonal eller sosial opprinnelse eiendom, fødsel eller "
"annet forhold.\n"
# Chinese
"并且不得因一人所属的国家或领土的政治的、行政的或者国际的地位之不同而有所区别,无论该领土是独立领土、托管领土、非自治领土或者处于其他任何主权受限制的情况之下。\n"
# Korean
"3. 모든 사람은 생명과 신체의 자유와 안전에 대한 권리를 가진다.\n"
# Dutch
"4. Slavernij en slavenhandel in iedere vorm zijn verboden.\n"
# Ukrainian
"5. Ніхто не повинен зазнавати тортур, або жорстокого, нелюдського, або такого, що "
"принижує його гідність, поводження і покарання."
),
(
"def fibonacci(limit: int) -> Iterator[int]:\n"
' """Yield Fibonacci numbers below `limit`."""\n'
" a, b = 0, 1\n"
" while a < limit:\n"
" yield a\n"
" a, b = b, a + b"
),
(
'{"user_id": 84213, "name": "Ana Sofía Ruiz", "locale": "es-MX",\n'
' "tags": ["premium", "beta"], "created_at": "2026-08-16T09:12:44Z",\n'
' "balance": 1234.56, "active": true, "referrer": null}'
),
(
"Revenue grew from $1,234,567.89 in 2023 to $2,847,193.05 in 2024, up 130.6%.\n"
"Order #A7X-99420 shipped on 2026-08-16 to 52.3676° N, 4.9041° E."
),
"🇳🇱 naïve café — “curly quotes” … ½ + ⅓ ≈ 0.83 — ambiguïteit, Straße, İstanbul",
]
EXAMPLE_LABELS = [
"Human rights, articles 1–5",
"Python",
"A JSON payload",
"Numbers, money and dates",
"Unicode that falls back to bytes",
]
with gr.Blocks(title="Claude Tokenizer") as demo:
gr.Markdown(INTRO)
gr.HTML(LINKS)
with gr.Row():
local_pick = gr.CheckboxGroup(
LOCAL_CHOICES, value=LOCAL_DEFAULT, label="Built-in tokenizers", scale=3
)
repo_pick = gr.Dropdown(
HF_SUGGESTIONS,
value=HF_DEFAULT,
multiselect=True,
allow_custom_value=True,
label="Hugging Face repos",
info="any repo id with a tokenizer.json — type your own and press enter",
scale=2,
)
with gr.Tab("Text"):
text_in = gr.Textbox(label="Text", lines=5, value=EXAMPLES[0], placeholder="Paste anything…")
gr.Examples(EXAMPLES, inputs=text_in, label="Examples", example_labels=EXAMPLE_LABELS)
text_note = gr.Markdown()
text_table = gr.HTML(label="Counts")
text_viz = gr.HTML(label="Segmentation")
# No Compare button: the table and the segmentation follow the textarea as you type.
text_inputs = [text_in, local_pick, repo_pick]
text_outputs = [text_table, text_note, text_viz]
demo.load(compare_text, text_inputs, text_outputs)
# Live-update as you type or change the lineup; `always_last` drops intermediate keystrokes.
for trigger in (text_in.change, local_pick.change, repo_pick.change):
trigger(
compare_text,
text_inputs,
text_outputs,
trigger_mode="always_last",
show_progress="minimal",
)
with gr.Tab("Dataset"):
with gr.Row():
ds_id = gr.Dropdown(
DATASET_SUGGESTIONS,
value="google/wmt24pp",
allow_custom_value=True,
label="Dataset",
info="any public dataset id",
)
ds_config = gr.Dropdown(label="Config", allow_custom_value=True)
ds_split = gr.Textbox("train", label="Split")
with gr.Row():
ds_column = gr.Dropdown(label="Text column", allow_custom_value=True)
ds_rows = gr.Slider(10, 1000, value=100, step=10, label="Rows")
with gr.Row():
peek_btn = gr.Button("Load columns")
ds_btn = gr.Button("Compare", variant="primary")
ds_note = gr.Markdown()
ds_table = gr.HTML(label="Counts")
# Hidden until there is data: an empty plot reads as a broken component.
ds_plot = gr.BarPlot(
x="tokenizer",
y="chars / token",
label="Characters per token — higher is cheaper",
visible=False,
)
ds_id.change(on_dataset_change, ds_id, ds_config)
demo.load(on_dataset_change, ds_id, ds_config)
for trigger in (peek_btn.click, ds_config.change):
trigger(peek_columns, [ds_id, ds_config, ds_split], [ds_column, ds_note])
ds_btn.click(
compare_dataset,
[ds_id, ds_config, ds_split, ds_column, ds_rows, local_pick, repo_pick],
[ds_table, ds_note, ds_plot],
)
if __name__ == "__main__":
# Pure CPU string work, so several visitors can be served at once; the default of 1 makes
# one person's large paste block everyone else's keystrokes.
demo.queue(default_concurrency_limit=4)
demo.launch(css=CSS, mcp_server=True)