"""Polygen tokenization demo -- Gradio Space. One upload box, any data. The demo auto-detects what you give it (image, audio, numeric table, text, or any other file) and tokenizes it through the polygen SDK, showing what the tokenizer buys beyond byte compression: a lossless round-trip, a compact queryable analytics archive (COARSE), and a per-segment anomaly signal. An unrecognized file still tokenizes -- its raw bytes are read as a 1-D signal -- so nothing errors. Local: ``POLYGEN_DEMO_LOCAL_MOCK=1 python app.py``. Deploy: set the ``POLYGEN_LICENSE`` secret on the Space. """ import license_bootstrap license_bootstrap.ensure_license() # must run before polygen is imported import column_readout # noqa: E402 import cost # noqa: E402 import demo_core # noqa: E402 import gradio as gr # noqa: E402 import numpy as np # noqa: E402 import plots # noqa: E402 # Datasent 2026 brand: Minsk purple primary, bright blue accent, lavender # light shades, white page. Mirrors docs/polygen_theme/custom.css and the # qwen2vl_polygen Space, so the demos share a house style. THEME = gr.themes.Soft( primary_hue="indigo", neutral_hue="slate", text_size=gr.themes.sizes.text_md, radius_size=gr.themes.sizes.radius_lg, ).set( block_background_fill="*background_fill_primary", block_border_width="1px", block_shadow="*shadow_drop_lg", button_primary_text_color="white", ) # Card styling is keyed off the ``.ds-summary-block`` class emitted by every # summary builder, so every result shares one look. BRAND_CSS = """ :root { --ds-minsk:#3c3475; --ds-blue:#4caaff; --ds-dark-3:#6464a0; } #ds-hero { text-align:center; padding:14px 0 4px 0; } #ds-hero img { height:34px; display:block; margin:0 auto 10px; } #ds-hero .eyebrow { font-size:0.8em; letter-spacing:0.18em; text-transform:uppercase; color:var(--ds-dark-3); margin-bottom:6px; } #ds-hero h1 { margin:0 0 8px 0; font-size:2.1em; line-height:1.1; color:var(--ds-minsk); } #ds-hero .tagline { color:#555579; font-size:1.05em; max-width:680px; margin:0 auto; } .ds-summary-block h3 { color:var(--ds-minsk); margin:6px 0 2px; } .ds-summary-block .ds-cards { display:flex; flex-wrap:wrap; gap:12px; margin:10px 0 14px; } .ds-summary-block .ds-card { flex:1 1 150px; border-radius:12px; padding:13px 15px; background:var(--block-background-fill, rgba(255,255,255,0.04)); border:1px solid rgba(148,163,184,0.28); border-top:3px solid var(--ds-blue); } .ds-summary-block .ds-card .k { font-size:0.72em; text-transform:uppercase; letter-spacing:0.07em; color:var(--ds-dark-3); } .ds-summary-block .ds-card .v { font-size:1.5em; font-weight:700; color:var(--ds-minsk); line-height:1.15; margin-top:5px; } .ds-summary-block .ds-card .s { font-size:0.8em; color:#555579; margin-top:3px; } .ds-summary-block .ds-note { color:#555579; font-size:0.96em; line-height:1.5; } /* Dark mode: brand deep-purple inverts to light periwinkle, muted text to a soft periwinkle-grey; cards inherit the theme block fill. Each .dark rule stands alone -- gradio's css= scoper only scopes the first selector in a comma group, so a grouped .dark selector silently never matches. */ .dark #ds-hero h1 { color:#c1cef7; } .dark #ds-hero .eyebrow { color:#b4bce0; } .dark #ds-hero .tagline { color:#b4bce0; } .dark .ds-summary-block h3 { color:#c1cef7; } .dark .ds-summary-block .ds-card .k { color:#b4bce0; } .dark .ds-summary-block .ds-card .v { color:#c1cef7; } .dark .ds-summary-block .ds-card .s { color:#b4bce0; } .dark .ds-summary-block .ds-note { color:#b4bce0; } """ HERO_HTML = """
Datasent
Datasent · Polygen

Tokenization Demo

Upload anything -- a numeric signal, an image, an audio clip, text, or any other file. The demo detects what it is and tokenizes it through one pipeline. Lossless when you want it, a tiny queryable archive when you don't.
""" def _rt_icon(r: dict) -> str: """Round-trip badge: a check at the quantization floor, else the error.""" peak = float(np.max(np.abs(r["original"]))) if np.size(r["original"]) else 1.0 return "✓" if r["max_err"] <= max(1e-2, 1e-4 * peak) else f"{r['max_err']:.1e}" def _summary(r: dict) -> str: anom = int(np.sum(r["sigma_r"] > np.mean(r["sigma_r"]) + 2 * np.std(r["sigma_r"]))) if r["segments"] > 1 else 0 return f"""

Results — {r["n"]:,} rows × {r["d"]} channel(s), {r["segments"]} windows

Lossless round-trip
{_rt_icon(r)}
max error {r["max_err"]:.1e}
Analytics archive
{r["coarse_ratio_vs_raw"]:.0f}×
smaller than raw, queryable
Lossless archive
{r["token_size"] / 1024:.1f} KB
vs gzip {r["gzip_size"] / 1024:.1f} KB
Anomalies flagged
{anom}
of {r["segments"]} windows
The lossless archive is exactly reversible and, for structured signals, lands below gzip (see the bars). The bigger win a byte codec cannot match is the compact analytics archive (the signal's shape at a fraction of the size), tokens that drop straight into ML as feature vectors, and a per-window anomaly score for free.
""".strip() def _image_summary(r: dict) -> str: psnr = r["psnr"] quality = "lossless" if not np.isfinite(psnr) else f"{psnr:.1f} dB" ratio = r["raw_size"] / r["token_size"] if r["token_size"] else 0.0 return f"""

Results — {r["img_h"]}×{r["img_w"]} image, {r["channels"]} channel(s)

Lossless round-trip
{_rt_icon(r)}
max error {r["max_err"]:.1e}
Analytics archive
{r["coarse_ratio_vs_raw"]:.0f}×
smaller than raw, queryable
Lossless archive
{r["token_size"] / 1024:.1f} KB
vs gzip {r["gzip_size"] / 1024:.1f} KB
Recovered image
{quality}
rebuilt from the tokens
Token size
{r["token_size"] / 1024:.1f} KB
vs {r["raw_size"] / 1024:.0f} KB raw ({ratio:.1f}×)
Compact features
{r["n_patches"]:,}
patch descriptors tokenized
Polygen turns the image into compact per-patch features and tokenizes them losslessly; the right panel is rebuilt from those tokens. Higher Detail keeps more coefficients (sharper image, larger token -- it can exceed gzip); the always-small artifact is the coefficients-only analytics archive. Same pipeline as every modality.
""".strip() def _pdf_summary(r: dict) -> str: psnr = r["psnr"] quality = "lossless" if not np.isfinite(psnr) else f"{psnr:.1f} dB" ratio = r["raw_size"] / r["token_size"] if r["token_size"] else 0.0 return f"""

Results — PDF page 1 of {r["pdf_pages"]}, rendered to {r["img_h"]}×{r["img_w"]}

Analytics archive
{r["coarse_ratio_vs_raw"]:.0f}×
smaller than the page, queryable
Lossless round-trip
{_rt_icon(r)}
max error {r["max_err"]:.1e}
Recovered page
{quality}
rebuilt from the tokens
Lossless token
{r["token_size"] / 1024:.1f} KB
{ratio:.1f}× smaller than the {r["raw_size"] / 1024:.0f} KB page
Compact features
{r["n_patches"]:,}
patch descriptors tokenized
A PDF is already a compressed container, so the demo renders the page to an image and tokenizes that. The payoff is a lossless, ML-ready token and a queryable analytics archive {r["coarse_ratio_vs_raw"]:.0f}× smaller than the page; the right panel is the page rebuilt from the tokens. Higher Detail keeps more coefficients (sharper page, larger token).
""".strip() def _audio_summary(r: dict) -> str: ratio = r["raw_size"] / r["token_size"] if r["token_size"] else 0.0 return f"""

Results — {r["duration_s"]:.1f}s at {r["sample_rate"]:,} Hz

Lossless round-trip
{_rt_icon(r)}
max error {r["max_err"]:.1e}
Analytics archive
{r["coarse_ratio_vs_raw"]:.0f}×
smaller than raw, queryable
Lossless archive
{r["token_size"] / 1024:.1f} KB
vs gzip {r["gzip_size"] / 1024:.1f} KB
Spectrogram
{r["n_frames"]}×{r["n_mels"]}
frames × bands
Token size
{r["token_size"] / 1024:.1f} KB
vs {r["raw_size"] / 1024:.0f} KB raw ({ratio:.1f}×)
The clip becomes a spectrogram -- a numeric matrix of per-frame energies -- which polygen tokenizes. The analytics archive is the coefficients only. Same pipeline as the numeric path.
""".strip() def _text_summary(r: dict) -> str: return f"""

Results — {r["n_chars"]:,} characters ({r["text_bytes"]:,} bytes)

Lossless round-trip
{_rt_icon(r)}
max error {r["max_err"]:.1e}
Analytics archive
{r["coarse_ratio_vs_raw"]:.0f}×
smaller than raw, queryable
Lossless archive
{r["token_size"] / 1024:.1f} KB
vs gzip {r["gzip_size"] / 1024:.1f} KB
Feature vector
{r["n_features"]:,}
dimensions
Active features
{r["nonzero"]:,}
non-zero weights
Token size
{r["token_size"] / 1024:.1f} KB
ML-ready numeric tokens
Text is feature extraction, not byte compression: the token encodes a fixed-width feature vector, so it is not smaller than gzip of the short raw text. The value is the consistent, ML-ready numeric representation (the same pipeline as every modality), plus the tiny coefficients-only analytics archive.
""".strip() def _bytes_summary(r: dict) -> str: ratio = r["raw_size"] / r["token_size"] if r["token_size"] else 0.0 return f"""

Results — {r["n_bytes"]:,} bytes, tokenized as a byte signal

Lossless round-trip
{_rt_icon(r)}
max error {r["max_err"]:.1e}
Analytics archive
{r["coarse_ratio_vs_raw"]:.0f}×
smaller than raw
Lossless archive
{r["token_size"] / 1024:.1f} KB
vs gzip {r["gzip_size"] / 1024:.1f} KB
Token size
{r["token_size"] / 1024:.1f} KB
vs {r["raw_size"] / 1024:.0f} KB raw ({ratio:.1f}×)
No recognized type, so polygen reads the raw bytes as a 1-D signal and tokenizes them anyway. Already-compressed files are high-entropy and won't shrink much -- that is honest, and it shows the pipeline runs on literally any input.
""".strip() # The Fitting radio's backing values stay generic ("auto" / "single") so no # basis-family name is shipped to the browser in the gradio config or API # schema. Map the UI value to the SDK basis mode here, at the server boundary. _FITTING_TO_BASIS = {"auto": "auto", "single": "chebyshev"} _SUMMARY_BUILDERS = { "image": _image_summary, "pdf": _pdf_summary, "audio": _audio_summary, "text": _text_summary, "bytes": _bytes_summary, } def _summary_for(r: dict) -> str: """Pick the summary card set for the detected modality (numeric default).""" return _SUMMARY_BUILDERS.get(r["modality"], _summary)(r) def _primary_plot_for(r: dict): """The modality-appropriate primary figure.""" modality = r["modality"] if modality in ("image", "pdf"): return plots.fig_image_pair(r) if modality == "audio": return plots.fig_spectrogram(r) if modality == "text": return plots.fig_text_features(r) if modality == "bytes": return plots.fig_bytes_signal(r) return plots.fig_numeric_overview(r) def _sizes_for(r: dict): """Stored-size comparison; PDFs drop the gzip baseline (not the point there).""" if r["modality"] == "pdf": return plots.fig_sizes_archive(r) return plots.fig_sizes(r) def run_any_ui(upload, sample, data_type, fitting, detail, auto_window, segment_length): """Detect or force the type, tokenize, and render the result.""" path = getattr(upload, "name", upload) if upload else None r = demo_core.run_any( path, sample, data_type=data_type, detail=detail, basis_mode=_FITTING_TO_BASIS.get(fitting, "auto"), auto_window=bool(auto_window), segment_length=int(segment_length), ) badge = f"**Detected:** {r['detected_label']}" # Reflect the actual window used so the greyed slider is not stuck at its default. window_update = gr.update(value=int(r["segment_length"])) if auto_window else gr.update() # ``r`` is returned last for the cost panel state; it is not a UI component. return ( badge, _summary_for(r), _primary_plot_for(r), _sizes_for(r), r["token_preview"], window_update, r, ) def _readout_ui(r): """Render the per-column honesty readout (numeric uploads only). The readout reads row-order structure along axis 0, which is only meaningful for numeric series. Image / audio / text / pdf results also carry 2-D original/coarse matrices (their representation), so gate on modality here rather than on array shape. """ if not r: return gr.update() if r.get("modality") != "numeric": return "" return column_readout.readout_html(r) def _cost_ui(r, units_per_month, corpus_units): """Render the cost-at-your-scale panel, softened when the data is unordered. Numeric uploads only: the projection counts numbers as ``raw_size / 4`` (float32), which holds for the numeric path but not for image / audio / text where ``raw_size`` is original file bytes -- and the validated cost line behind the panel is row-ordered numeric signals in the first place. """ if not r: return gr.update() if r.get("modality") != "numeric": return "" structured = column_readout.mostly_structured(r) return cost.panel_html(r, units_per_month, corpus_units, structured) def _detail_update(data_type, fitting): """Build the gr.update for the Detail slider. The chosen type sets the range/preset; fitting gates the numeric/bytes degree to Single-basis mode. """ if data_type == "auto": return gr.update( interactive=False, label="Detail (auto per detected type)", info="Set automatically once the file type is detected", ) spec = demo_core.DETAIL_SPECS[data_type] active, info = True, "" if data_type in ("numeric", "bytes"): active = fitting == "single" info = "Degree for the fixed-basis fit" if active else "Only affects the Single fixed basis mode" return gr.update( minimum=spec["min"], maximum=spec["max"], step=spec["step"], value=spec["preset"], label=spec["label"], info=info, interactive=active, ) _TOKEN_EXPLAINER = ( "**The polygen token** is the actual artifact the SDK stores or transmits " "instead of the raw data: a compact binary container -- a format header " "(the `DSL3` magic in the first four bytes) followed by Zstd-compressed " "model coefficients and residuals. It decodes back with no loss " "(prediction plus residual). The hex below is the start of that token." ) def build_ui() -> gr.Blocks: with gr.Blocks(title="Datasent's Polygen Tokenization Demo", theme=THEME, css=BRAND_CSS) as ui: gr.HTML(HERO_HTML) with gr.Row(): with gr.Column(scale=1): upload = gr.File( label="Upload anything -- image, PDF, audio, CSV, text, or any file", ) sample = gr.Dropdown( choices=demo_core.SAMPLE_CHOICES, value=demo_core.SAMPLE_CHOICES[0], label="...or pick a sample", ) data_type = gr.Dropdown( choices=[ ("Auto-detect", "auto"), ("Image", "image"), ("PDF", "pdf"), ("Audio", "audio"), ("Text", "text"), ("Numeric table", "numeric"), ("Raw bytes", "bytes"), ], value="auto", label="Data type", info="Auto-detect, or declare the type for its best-tuned config", ) fitting = gr.Radio( choices=[("MDL adaptive selection", "auto"), ("Single fixed basis", "single")], value="auto", label="Fitting", ) detail = gr.Slider( 1, 10, value=5, step=1, label="Detail (auto per detected type)", info="Set automatically once the file type is detected", interactive=False, ) auto_window = gr.Checkbox( value=True, label="Auto window length", info="Pick the best window size automatically", ) seg = gr.Slider( 256, 8192, value=4096, step=256, label="Window length", info="Auto-chosen per input; uncheck Auto to set it yourself", interactive=False, ) go = gr.Button("Tokenize", variant="primary") with gr.Column(scale=2): detected = gr.Markdown() summary = gr.HTML() # show_label=False: each figure already carries a descriptive matplotlib # title ("Reconstruction: ...", "Stored size: ..."). The gradio Plot label # floats a pill over the top-left of the canvas, covering that title, so # the in-figure title is the single, non-overlapping heading. with gr.Row(): primary = gr.Plot(show_label=False) sizes = gr.Plot(show_label=False) gr.Markdown(_TOKEN_EXPLAINER) token = gr.Code(label="Polygen token (hex preview)", interactive=False) # Per-column honesty readout + ordering guard (numeric uploads only). readout = gr.HTML() # Cost at your scale: project THIS file's measured compression onto the # visitor's own volume (the calculator pattern). Recomputes on tokenize and # whenever a volume input changes, without re-tokenizing. result_state = gr.State(None) with gr.Row(): volume = gr.Number( value=1e6, label="Files per month", info="Your monthly volume of files like this one", ) corpus = gr.Number( value=5e7, label="Files retained", info="How many such files you keep (the stored corpus)", ) cost_html = gr.HTML() inputs = [upload, sample, data_type, fitting, detail, auto_window, seg] outputs = [detected, summary, primary, sizes, token, seg, result_state] # Auto on -> the window slider is inert; grey it out for clarity. auto_window.change(lambda a: gr.update(interactive=not a), auto_window, seg) # The Detail slider re-ranges per declared type (and gates on Fitting). data_type.change(_detail_update, [data_type, fitting], detail) fitting.change(_detail_update, [data_type, fitting], detail) cost_inputs = [result_state, volume, corpus] ( go.click(run_any_ui, inputs, outputs) .then(_readout_ui, result_state, readout) .then(_cost_ui, cost_inputs, cost_html) ) ( ui.load(run_any_ui, inputs, outputs) .then(_readout_ui, result_state, readout) .then(_cost_ui, cost_inputs, cost_html) ) # Re-price on volume change without re-running the tokenizer. volume.change(_cost_ui, cost_inputs, cost_html) corpus.change(_cost_ui, cost_inputs, cost_html) return ui # HuggingFace Spaces auto-launches a module-level ``demo``. Building it at # import requires a valid license: the POLYGEN_LICENSE secret on the Space, # or POLYGEN_DEMO_LOCAL_MOCK=1 locally (handled by ensure_license above). demo = build_ui() if __name__ == "__main__": # ssr_mode=False: gradio 5 SSR renders blank behind HF Spaces' proxy. demo.launch(ssr_mode=False)