File size: 9,241 Bytes
0b999a6 e841b14 0b999a6 e841b14 0b999a6 e841b14 0b999a6 e841b14 0b999a6 e841b14 0b999a6 e841b14 0b999a6 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 | """© KAND CA 2026 - Arabic tokenizer comparison Space.
A tokenizer is the cheapest thing to get wrong in an Arabic pipeline and the
hardest to see. Every model card quotes parameters and context length; almost
none tell you that the same Arabic paragraph costs 52% more context on one
tokenizer than another, which is a 52% tax on every prompt, every document and
every embedding you will ever run through it.
This puts the number on screen. Paste Arabic - MSA, dialect, or code-switched -
and see how many tokens each tokenizer actually spends on it.
The interesting comparison is not big-vocab vs small-vocab. Mistral and Emhotob
both have ~32K entries; Emhotob spends all of them on Arabic and Mistral spends
almost none, and the gap that opens up is the whole point.
"""
import html
import statistics
import gradio as gr
from transformers import AutoTokenizer
TOKENIZERS = [
("Emhotob 32K (Arabic-only)", "oddadmix/50M-2048-Emhotob", "Arabic-only"),
("Gemma-4", "google/gemma-4-31B-it", "multilingual"),
("Qwen3.6", "Qwen/Qwen3.6-27B", "multilingual"),
("Qwen3.5", "Qwen/Qwen3.5-4B", "multilingual"),
("Mistral-7B v0.3", "mistralai/Mistral-7B-v0.3", "western"),
("GPT-2", "openai-community/gpt2", "western"),
]
EXAMPLES = {
"فصحى — MSA news": (
"أعلنت وزارة الاقتصاد أن معدل النمو المتوقع خلال العام المقبل سيبلغ نحو "
"أربعة في المئة، مدفوعاً بارتفاع الصادرات غير النفطية وتحسن أداء قطاع "
"السياحة، فيما أشار التقرير إلى أن الاستثمارات الأجنبية المباشرة سجلت "
"زيادة ملحوظة مقارنة بالفترة نفسها من العام الماضي."),
"مصري — Egyptian": (
"يا جماعة أنا رايح السوق دلوقتي عشان أجيب شوية حاجات للبيت، لو محتاجين "
"حاجة قولولي بسرعة قبل ما أنزل. الأسعار غليت أوي الفترة دي وبقى لازم "
"الواحد يحسب حسابه كويس قبل ما يشتري أي حاجة."),
"مغربي — Moroccan Darija": (
"واش نتا فاهم شنو كايقع؟ الأسعار طلعات بزاف هاد الشهر وكلشي كايشتكي. "
"غادي نمشي للسوق دابا باش نشوف شنو كاين، ولكن ما كنظنش غادي نلقى شي حاجة "
"رخيصة. الله يسهل علينا."),
"خليجي — Gulf": (
"شلونك؟ أنا بروح الدوام بدري اليوم لأن عندي اجتماع مهم مع المدير. "
"بعدين بمر على المحل عشان أشتري أغراض البيت، وإذا خلصت بدري بجيك."),
"Code-switched": (
"الـ deployment اتعمل امبارح على الـ production server بس فيه issue في "
"الـ latency، محتاجين نعمل profiling للـ database queries عشان نشوف "
"الـ bottleneck فين بالظبط."),
"English (control)": (
"The ministry announced that the expected growth rate for the coming year "
"will reach approximately four percent, driven by rising non-oil exports "
"and improved performance in the tourism sector."),
}
_CACHE = {}
def get(repo):
if repo not in _CACHE:
_CACHE[repo] = AutoTokenizer.from_pretrained(repo)
return _CACHE[repo]
PALETTE = ["#dbeafe", "#fef3c7", "#dcfce7", "#fae8ff", "#ffe4e6", "#e0f2fe"]
def render_tokens(tk, ids):
"""Colour each token so the segmentation is visible, not just counted.
Tokens are decoded ONE ID AT A TIME rather than read off
`convert_ids_to_tokens`. For a byte-level BPE - which Emhotob, Gemma, Qwen
and GPT-2 all are - that method returns the byte-mangled form, so Arabic
comes back as 'أعÙĦÙĨت' instead of 'أعلنت'. Decoding per id gives the
real characters, which is the whole point of showing the split.
"""
out = []
for i, tid in enumerate(ids):
s = tk.decode([tid])
lead = s.startswith(" ")
s = html.escape(s.strip()) or "␣"
out.append(
f'<span style="background:{PALETTE[i % len(PALETTE)]};'
f'padding:3px 5px;margin:2px;border-radius:4px;'
f'display:inline-block;color:#111;font-size:15px;'
f'border-left:{"3px solid #94a3b8" if lead else "0"}">{s}</span>')
return ('<div dir="rtl" style="line-height:2.4;direction:rtl;'
'text-align:right;padding:10px;background:#fafafa;'
'border-radius:8px;border:1px solid #e5e7eb">'
+ "".join(out) + "</div>")
def compare(text, show_for):
text = (text or "").strip()
if not text:
raise gr.Error("اكتب أو الصق نصاً عربياً أولاً. / Enter some text first.")
n_words = len(text.split())
n_chars = len(text)
rows, counts = [], {}
for name, repo, kind in TOKENIZERS:
try:
tk = get(repo)
ids = tk(text, add_special_tokens=False)["input_ids"]
counts[name] = len(ids)
rows.append([name, kind, f"{tk.vocab_size:,}", len(ids),
round(len(ids) / max(n_words, 1), 3),
round(n_chars / max(len(ids), 1), 2)])
except Exception as e:
rows.append([name, kind, "—", None, None, None])
base = counts.get("Emhotob 32K (Arabic-only)")
for r in rows:
r.append(round(r[3] / base, 2) if (base and r[3]) else None)
ok = [r for r in rows if r[3]]
ok.sort(key=lambda r: r[3])
best, worst = ok[0], ok[-1]
summary = (
f"### {n_words} كلمة · {n_chars:,} حرف\n\n"
f"**{best[0]}** is most efficient at **{best[3]:,} tokens** "
f"({best[4]} tok/word). **{worst[0]}** needs **{worst[3]:,}** "
f"— **{round(worst[3]/best[3], 2)}×** as many for the same text.\n\n"
f"On a 128K context window that difference is "
f"**{int(128000/best[4]) - int(128000/worst[4]):,} fewer words** of room.")
tk = get(dict((n, r) for n, r, _ in TOKENIZERS)[show_for])
ids = tk(text, add_special_tokens=False)["input_ids"]
return rows, summary, render_tokens(tk, ids)
with gr.Blocks(title="Arabic Tokenizer Comparison") as demo:
gr.Markdown(
"# 🔤 Arabic Tokenizer Comparison\n"
"### كم رمزاً يكلّفك النص العربي؟\n\n"
"The same Arabic paragraph can cost **50% more context** on one tokenizer "
"than another. That is a tax on every prompt, document and embedding you "
"run — and no model card mentions it. Paste Arabic below and see.\n\n"
"Note that **Mistral and Emhotob both have ~32K vocabularies**. The gap "
"between them is not vocabulary *size*, it is what the vocabulary is "
"*spent on*.")
with gr.Row():
with gr.Column(scale=3):
text = gr.Textbox(label="النص / Text", lines=8, rtl=True,
text_align="right", value=EXAMPLES["فصحى — MSA news"])
gr.Examples(examples=[[v] for v in EXAMPLES.values()], inputs=[text],
example_labels=list(EXAMPLES), label="أمثلة / Examples")
with gr.Column(scale=2):
summary = gr.Markdown()
show_for = gr.Dropdown([n for n, _, _ in TOKENIZERS],
value="Emhotob 32K (Arabic-only)",
label="Show token split for")
run = gr.Button("قارِن / Compare", variant="primary")
table = gr.Dataframe(
headers=["tokenizer", "kind", "vocab", "tokens", "tok/word", "chars/tok", "×Emhotob"],
datatype=["str", "str", "str", "number", "number", "number", "number"],
label="fewer tokens = better", wrap=True)
viz = gr.HTML(label="token split — a grey edge marks a token that begins with a space")
gr.Markdown(
"---\n"
"**Emhotob 32K** is the Arabic-only byte-level BPE behind "
"[Nawah](https://huggingface.co/oddadmix/Nawah-50M-RAG-Support-2K) and "
"[50M-2048-Emhotob](https://huggingface.co/oddadmix/50M-2048-Emhotob). "
"Every one of its 32,000 entries is spent on Arabic, which is why it "
"beats vocabularies 8× its size on Arabic text — and why it does *worse* "
"on English, which it was never meant to handle.\n\n"
"Measured on MSA it reaches ~1.39 tokens/word against Gemma-4's ~2.11. "
"On dialect its lead narrows — dialectal orthography is where an "
"MSA-trained vocabulary is weakest.\n\n"
"© KAND CA 2026 — PROJECT NAWAH")
run.click(compare, [text, show_for], [table, summary, viz])
text.submit(compare, [text, show_for], [table, summary, viz])
show_for.change(compare, [text, show_for], [table, summary, viz])
if __name__ == "__main__":
demo.queue(max_size=24).launch(theme=gr.themes.Soft(primary_hue="teal"))
|