import html import os import re from collections.abc import Iterator from pathlib import Path from threading import Thread PREVIEW = os.getenv("TWIL_UI_PREVIEW") == "1" if not PREVIEW: import spaces import torch from transformers import AutoModelForCausalLM, AutoTokenizer, TextIteratorStreamer else: class spaces: # type: ignore[no-redef] @staticmethod def GPU(duration=None, **kwargs): def decorator(fn): return fn return decorator import gradio as gr from latex2mathml.converter import convert as latex_to_mathml MODEL_ID = "webAI-Official/TwIL-LM3-Pro" MAX_INPUT_TOKEN_LENGTH = int(os.getenv("MAX_INPUT_TOKEN_LENGTH", "8192")) ASSETS = Path(__file__).resolve().parent / "brand" AVATAR = ASSETS / "avatar.svg" THINK_OPEN = chr(60) + "think" + chr(62) THINK_CLOSE = chr(60) + chr(47) + "think" + chr(62) _THINK_RE = re.compile( re.escape(THINK_OPEN) + r".*?(" + re.escape(THINK_CLOSE) + r"|$)", re.DOTALL, ) SYSTEM_PROMPT = ( "You are TwIL-LM3-Pro, a formal-logic reasoning model created by webAI Intelligence Lab. " "You specialise in entailment, first-order logic, semantic parsing, Lean, and proof critique. " f"Work through the problem in a {THINK_OPEN} block, then give a concise, precise answer. " "Format all LaTeX as Markdown math: use $...$ inline and $$...$$ for display equations." ) LATEX_SYMBOLS = ( (r"\Longleftrightarrow", "⟺"), (r"\Longrightarrow", "⟹"), (r"\Longleftarrow", "⟸"), (r"\leftrightarrow", "↔"), (r"\Leftrightarrow", "⇔"), (r"\rightarrow", "→"), (r"\Rightarrow", "⇒"), (r"\leftarrow", "←"), (r"\Leftarrow", "⇐"), (r"\nrightarrow", "↛"), (r"\nleftarrow", "↚"), (r"\therefore", "∴"), (r"\because", "∵"), (r"\emptyset", "∅"), (r"\subseteq", "⊆"), (r"\supseteq", "⊇"), (r"\subset", "⊂"), (r"\supset", "⊃"), (r"\notin", "∉"), (r"\models", "⊨"), (r"\nvdash", "⊬"), (r"\vdash", "⊢"), (r"\forall", "∀"), (r"\exists", "∃"), (r"\neg", "¬"), (r"\lnot", "¬"), (r"\land", "∧"), (r"\wedge", "∧"), (r"\lor", "∨"), (r"\vee", "∨"), (r"\oplus", "⊕"), (r"\implies", "⇒"), (r"\iff", "⇔"), (r"\neq", "≠"), (r"\leq", "≤"), (r"\geq", "≥"), (r"\infty", "∞"), (r"\in", "∈"), (r"\bot", "⊥"), (r"\top", "⊤"), (r"\to", "→"), ) CHIPS = [ ( "Analyze an argument", "Analyze this argument: identify the premises and the conclusion, say whether it is valid, and name any fallacy.\n\n", ), ( "Check an entailment", "Do these premises entail the conclusion? Give a formal proof or a counterexample.\n\nPremise 1: \nPremise 2: \nConclusion: ", ), ( "Translate to FOL", "Translate into first-order logic: Every student who studies hard passes at least one exam.", ), ] SHOWCASE_TEMPLATE = ( "Analyze this argument: extract the premises and the conclusion, translate them " "into first-order logic, state whether the argument is valid, and name any fallacy.\n\n" 'Argument: "{argument}"' ) SHOWCASE = [ { "kicker": "Viral tweet", "quote": "The streets are wet this morning, and rain always makes the streets wet. So it obviously rained last night.", "hint": "Can wet streets prove rain?", }, { "kicker": "School board speech", "quote": "If we let students retake this one exam, soon they will demand to retake every exam, and before long no grade at this school will mean anything.", "hint": "Does one retake doom every grade?", }, { "kicker": "Supplement ad", "quote": "Our formula is 100% natural, and nature knows best. That means it is completely safe for your body.", "hint": "Does natural imply safe?", }, { "kicker": "Campaign rally", "quote": "Either you support this bill or you don't care about public safety. You oppose the bill, so you don't care about public safety.", "hint": "Are those really the only options?", }, { "kicker": "Toothpaste commercial", "quote": "4 out of 5 dentists we surveyed recommend SparkleDent, and experts know best. You should switch to SparkleDent today.", "hint": "What is the survey hiding?", }, { "kicker": "HR policy memo", "quote": "Every employee who completes the security training receives a certificate. Sam completed the security training. Therefore Sam receives a certificate.", "hint": "One of these six actually holds up.", }, ] SHOWCASE_INTRO = """

Can TwIL spot the flaw?

Six arguments from the wild. One click and TwIL extracts the premises, translates them to first-order logic, and rules on validity. Five are broken — one holds up.

""" def _card_html(case: dict) -> str: return ( '
' f'
{case["kicker"]}
' f'
“{case["quote"]}”
' f'
{case["hint"]}
' "
" ) WAITLIST_URL = "https://www.webai.com/waitlist" HEADER = f"""
webAI
webAI
TwIL-LM3-Pro · Intelligence Lab
Get early access
""" HEAD = """ """ PLACEHOLDER = f"""

What should we work on?

TwIL-LM3-Pro is webAI's formal-logic model — entailment, FOL, Lean, rule induction.

TwIL and the full family of webAI models are coming to the webAI app. Be the first to try expert intelligence on everyday devices.

Join the waitlist
""" CSS = """ :root { --weba-canvas: #f4f4f4; --weba-fg: #161616; --weba-muted: #737373; --weba-secondary: #e8e8e8; --weba-border: rgba(22, 22, 22, 0.1); --weba-composer: #ffffff; --weba-primary: #232323; --weba-primary-fg: #fafafa; } .dark { --weba-canvas: #232323; --weba-fg: #fafafa; --weba-muted: #a8a8a8; --weba-secondary: #323232; --weba-border: rgba(255, 255, 255, 0.1); --weba-composer: #161616; --weba-primary: #e8e8e8; --weba-primary-fg: #232323; } * { box-shadow: none !important; text-shadow: none !important; } html, body, #root, .gradio-container, .gradio-container > .main, .gradio-container .contain, .fillable, .gradio-container .column, .gradio-container .row, .contain, .wrapper { background: var(--weba-canvas) !important; box-shadow: none !important; filter: none !important; } /* Never paint over content while a job is running: the status tracker overlay must stay transparent or the chat looks blank mid-generation. */ .gradio-container [data-testid="status-tracker"], .gradio-container .wrap.default { background: transparent !important; } html, body, .gradio-container, .gradio-container > .main, .fillable, .contain, .app { height: 100% !important; min-height: 100vh !important; max-height: 100vh !important; max-width: none !important; margin: 0 !important; overflow: hidden !important; } .gradio-container, .main.fillable, .contain { display: flex !important; flex-direction: column !important; padding: 0 !important; } .gradio-container { font-family: ui-sans-serif, system-ui, -apple-system, "Segoe UI", sans-serif !important; color: var(--weba-fg) !important; } footer, .footer, .built-with, .settings, .settings-bar, .show-api, .show-api-divider, .divider { display: none !important; } .contain > .column:has(> #app-shell) { min-height: 0 !important; height: 100% !important; overflow: hidden !important; } /* Bound the shell to the viewport so the chat flexes into the leftover space; on very short screens the shell itself scrolls instead of clipping. */ #app-shell { flex: 1 1 auto !important; height: 100vh !important; height: 100dvh !important; max-height: 100dvh !important; min-height: 0 !important; display: flex !important; flex-direction: column !important; padding: 16px 24px 12px !important; box-sizing: border-box !important; overflow-x: hidden !important; overflow-y: auto !important; background: var(--weba-canvas) !important; } #app-shell > .gap, #app-shell > div { background: transparent !important; flex-grow: 0 !important; height: auto !important; min-height: 0 !important; } #app-shell > #twil-chat, #app-shell > .block:has(#twil-chat), #app-shell > div:has(#twil-chat) { flex: 1 1 auto !important; min-height: 0 !important; height: auto !important; } .twil-header { display: flex; align-items: center; gap: 12px; padding: 4px 0 8px; flex-shrink: 0; } .twil-mark { width: 28px; height: 28px; border-radius: 6px; } .twil-wordmark { font-weight: 600; font-size: 15px; letter-spacing: -0.02em; line-height: 1.2; color: var(--weba-fg); } .twil-product { font-size: 12px; color: var(--weba-muted); line-height: 1.3; } .twil-cta { display: inline-flex; align-items: center; height: 34px; padding: 0 18px; border-radius: 999px; background: var(--weba-primary); color: var(--weba-primary-fg) !important; font-size: 13px; font-weight: 500; text-decoration: none !important; border: none; white-space: nowrap; transition: opacity 0.15s ease; } .twil-cta:hover { opacity: 0.85; } .twil-cta-header { /* Keep clear of the floating HF Space pill in the top-right corner */ margin-left: 20px; height: 30px; padding: 0 14px; font-size: 12px; } .twil-waitlist { margin-top: 28px; display: flex; flex-direction: column; align-items: center; gap: 12px; } .twil-waitlist p { margin: 0; max-width: 30rem; font-size: 13px; line-height: 1.5; color: var(--weba-muted); } .twil-empty { min-height: 100%; box-sizing: border-box; display: flex; flex-direction: column; align-items: center; justify-content: safe center; text-align: center; padding: 12px; } #twil-chat .bubble-wrap:has(.placeholder-content) { padding-top: 0 !important; } @media (max-height: 820px), (max-width: 640px) { #twil-chat .twil-empty { padding: 0 12px; } #twil-chat .twil-cube { width: 56px !important; height: 56px !important; margin: 0 0 12px !important; } #twil-chat .twil-empty h1 { font-size: 22px !important; margin: 0 0 6px !important; } #twil-chat .twil-empty p { font-size: 14px; } #twil-chat .twil-waitlist { margin-top: 14px; } #twil-chat .twil-waitlist p { display: none; } } .twil-cube { width: 96px; height: 96px; margin-bottom: 24px; } .twil-empty h1 { font-size: 28px; font-weight: 500; letter-spacing: -0.03em; margin: 0 0 8px; color: var(--weba-fg); } .twil-empty p { margin: 0; max-width: 28rem; font-size: 15px; line-height: 1.5; color: var(--weba-muted); } #twil-chat, #twil-chat > .wrapper, #twil-chat .bubble-wrap, #twil-chat .message-wrap, #twil-chat .placeholder-content { background: transparent !important; border: none !important; box-shadow: none !important; min-height: 0 !important; } #twil-chat { max-width: 768px !important; width: 100% !important; height: auto !important; max-height: none !important; min-height: 200px !important; margin: 0 auto !important; flex: 1 1 0 !important; display: flex !important; flex-direction: column !important; overflow: hidden !important; } #twil-chat > .wrapper { flex: 1 1 auto !important; height: 100% !important; min-height: 0 !important; display: flex !important; flex-direction: column !important; overflow: hidden !important; } #twil-chat .bubble-wrap, #twil-chat .messages, #twil-chat [role="log"] { flex: 1 1 auto !important; height: 100% !important; max-height: 100% !important; min-height: 0 !important; overflow-x: hidden !important; overflow-y: auto !important; overscroll-behavior: contain; scrollbar-gutter: stable; padding-right: 8px !important; } #twil-chat .bubble-wrap::-webkit-scrollbar, #twil-chat .messages::-webkit-scrollbar, #twil-chat [role="log"]::-webkit-scrollbar { width: 8px; } #twil-chat .bubble-wrap::-webkit-scrollbar-thumb, #twil-chat .messages::-webkit-scrollbar-thumb, #twil-chat [role="log"]::-webkit-scrollbar-thumb { background: color-mix(in srgb, var(--weba-muted) 45%, transparent); border-radius: 999px; } #twil-chat .message-wrap, #twil-chat .placeholder-content { flex: 0 0 auto !important; } #twil-chat .message.user, #twil-chat .user, #twil-chat .bubble.user { background: var(--weba-secondary) !important; color: var(--weba-fg) !important; border: none !important; box-shadow: none !important; border-radius: 12px !important; } #twil-chat .message.bot, #twil-chat .bot, #twil-chat .bubble.bot { background: transparent !important; border: none !important; box-shadow: none !important; } #twil-chat [aria-label="Delete"], #twil-chat [aria-label="Copy"], #twil-chat [aria-label="Share"], #twil-chat .message-buttons { display: none !important; } #twil-chat details, #twil-chat .thought, #twil-chat .md.thought { background: transparent !important; border: none !important; box-shadow: none !important; color: var(--weba-muted) !important; } #twil-chat math { font-size: 1.15em; } #twil-chat math[display="block"] { margin: 0.4em 0; } #twil-chat details > summary { cursor: pointer !important; user-select: none; color: var(--weba-muted) !important; } #twil-chat details > summary:hover { color: var(--weba-fg) !important; } #composer-row { max-width: 768px !important; width: 100% !important; margin: 8px auto 0 !important; display: flex !important; flex-direction: row !important; flex-wrap: nowrap !important; align-items: center !important; gap: 10px !important; flex: 0 0 auto !important; min-height: 56px !important; background: transparent !important; } #composer-row > .block:has(#twil-input), #twil-input { flex: 1 1 0% !important; min-width: 0 !important; width: auto !important; height: 52px !important; min-height: 52px !important; max-height: 52px !important; background: var(--weba-composer) !important; border: 1px solid var(--weba-border) !important; border-radius: 999px !important; box-shadow: none !important; overflow: hidden !important; } #twil-input label, #twil-input .input-container { display: flex !important; align-items: stretch !important; width: 100% !important; height: 100% !important; min-height: 52px !important; margin: 0 !important; padding: 0 !important; border: none !important; background: transparent !important; box-sizing: border-box !important; } #twil-input textarea, #twil-input input, #twil-input .scroll-hide, #twil-input [data-testid="textbox"] { display: block !important; background: transparent !important; color: var(--weba-fg) !important; border: none !important; box-shadow: none !important; width: 100% !important; height: 52px !important; min-height: 52px !important; max-height: 52px !important; padding: 16px 18px !important; font-size: 16px !important; line-height: 20px !important; overflow-y: auto !important; overflow-x: hidden !important; resize: none !important; white-space: pre-wrap !important; field-sizing: fixed !important; pointer-events: auto !important; } #twil-input button, #twil-input .show-count, #twil-input .icon-button, #twil-input .sr-only, #twil-input span[data-testid="block-info"] { display: none !important; } #composer-row > .form, #composer-row > .block:has(#send-btn) { flex: 0 0 40px !important; width: 40px !important; min-width: 40px !important; max-width: 40px !important; background: transparent !important; border: none !important; padding: 0 !important; box-shadow: none !important; } #send-btn { flex: 0 0 40px !important; width: 40px !important; min-width: 40px !important; max-width: 40px !important; height: 40px !important; border-radius: 999px !important; background: var(--weba-secondary) !important; color: var(--weba-fg) !important; border: none !important; box-shadow: none !important; align-self: center !important; padding: 0 !important; } #chips-row { max-width: 768px !important; width: 100% !important; margin: 0 auto !important; display: flex !important; flex-wrap: wrap !important; justify-content: center !important; gap: 8px !important; flex-shrink: 0 !important; background: transparent !important; padding: 4px 0 8px !important; } #chips-row button { height: 32px !important; padding: 0 14px !important; border-radius: 999px !important; border: 1px solid var(--weba-border) !important; background: transparent !important; color: var(--weba-muted) !important; box-shadow: none !important; font-size: 13px !important; font-weight: 400 !important; white-space: nowrap !important; flex: 0 1 auto !important; width: auto !important; min-width: 0 !important; } #chips-row button:hover { background: var(--weba-secondary) !important; color: var(--weba-fg) !important; } @keyframes twil-pulse { 0%, 100% { opacity: 0.4; } 50% { opacity: 1; } } #twil-chat .pending, #twil-chat [data-status="pending"] { animation: twil-pulse 1.1s ease-in-out infinite; } #params-box, #params-box.block, #params-box .wrap, #params-box .styler, #params-box .gap, #accordion-content { max-width: 768px !important; margin: 4px auto 0 !important; background: transparent !important; border: none !important; box-shadow: none !important; flex: 0 0 auto !important; flex-grow: 0 !important; height: auto !important; min-height: 0 !important; max-height: none !important; overflow: visible !important; } #params-box .label-wrap, #params-box .icon { color: var(--weba-muted) !important; font-size: 12px !important; box-shadow: none !important; background: transparent !important; border: none !important; } .twil-disclaimer { margin: 8px 0 0; text-align: center; font-size: 11px; color: var(--weba-muted); flex-shrink: 0; } /* Tabs shell: keep the chat layout filling the viewport */ #twil-tabs { flex: 1 1 auto !important; min-height: 0 !important; display: flex !important; flex-direction: column !important; background: transparent !important; border: none !important; box-shadow: none !important; } #twil-tabs > .tab-wrapper, #twil-tabs .tab-container, #twil-tabs .tab-nav { flex: 0 0 auto !important; background: transparent !important; border: none !important; box-shadow: none !important; justify-content: center !important; } #twil-tabs .tab-container button, #twil-tabs .tab-nav button { background: transparent !important; border: none !important; border-bottom: 2px solid transparent !important; border-radius: 0 !important; box-shadow: none !important; color: var(--weba-muted) !important; font-size: 13px !important; padding: 6px 14px !important; } #twil-tabs .tab-container button.selected, #twil-tabs .tab-nav button.selected { color: var(--weba-fg) !important; border-bottom-color: var(--weba-fg) !important; } #twil-tabs .tabitem { flex: 1 1 auto !important; min-height: 0 !important; overflow-x: hidden !important; overflow-y: auto !important; background: transparent !important; border: none !important; box-shadow: none !important; padding: 0 !important; } #tab-chat > .gap, #tab-chat > .column, #tab-chat, #tab-chat .column:has(> div > #twil-chat) { height: 100% !important; min-height: 0 !important; } #tab-chat > .gap, #tab-chat > .column { display: flex !important; flex-direction: column !important; } #tab-chat .block:has(#twil-chat), #tab-chat div:has(> #twil-chat) { flex: 1 1 auto !important; min-height: 0 !important; height: auto !important; } #tab-showcase { overflow-y: auto !important; } /* Showcase cards */ .twil-showcase-intro { max-width: 768px; margin: 8px auto 4px; text-align: center; } .twil-showcase-intro h2 { font-size: 22px; font-weight: 500; letter-spacing: -0.02em; margin: 0 0 6px; color: var(--weba-fg); } .twil-showcase-intro p { margin: 0 0 8px; font-size: 14px; line-height: 1.5; color: var(--weba-muted); } .twil-card-row { max-width: 900px !important; width: 100% !important; margin: 6px auto !important; display: flex !important; gap: 12px !important; align-items: stretch !important; background: transparent !important; } .twil-card { background: var(--weba-composer) !important; border: 1px solid var(--weba-border) !important; border-radius: 16px !important; box-shadow: none !important; padding: 16px !important; gap: 10px !important; display: flex !important; flex-direction: column !important; justify-content: space-between !important; min-height: 170px !important; } .twil-card > * { background: transparent !important; } .twil-card-kicker { font-size: 11px; font-weight: 600; letter-spacing: 0.06em; text-transform: uppercase; color: var(--weba-muted); margin-bottom: 8px; } .twil-card-quote { font-size: 14px; line-height: 1.45; color: var(--weba-fg); margin-bottom: 8px; } .twil-card-hint { font-size: 12px; color: var(--weba-muted); font-style: italic; } .twil-card-btn { height: 32px !important; border-radius: 999px !important; border: 1px solid var(--weba-border) !important; background: var(--weba-secondary) !important; color: var(--weba-fg) !important; box-shadow: none !important; font-size: 13px !important; font-weight: 500 !important; align-self: flex-start !important; padding: 0 16px !important; width: auto !important; min-width: 0 !important; flex: 0 0 auto !important; } .twil-card-btn:hover { background: var(--weba-fg) !important; color: var(--weba-canvas) !important; } """ THEME = gr.themes.Base( primary_hue="zinc", secondary_hue="zinc", neutral_hue="zinc", font=[gr.themes.GoogleFont("Inter"), "ui-sans-serif", "system-ui", "sans-serif"], radius_size=gr.themes.sizes.radius_lg, ).set( body_background_fill="#f4f4f4", body_background_fill_dark="#232323", body_text_color="#161616", body_text_color_dark="#fafafa", background_fill_primary="transparent", background_fill_primary_dark="transparent", background_fill_secondary="#e8e8e8", background_fill_secondary_dark="#323232", border_color_primary="rgba(22,22,22,0.1)", border_color_primary_dark="rgba(255,255,255,0.1)", block_background_fill="transparent", block_background_fill_dark="transparent", block_border_width="0px", block_shadow="none", block_shadow_dark="none", shadow_drop="none", shadow_drop_lg="none", input_background_fill="#ffffff", input_background_fill_dark="#161616", input_border_color="rgba(22,22,22,0.1)", input_border_color_dark="rgba(255,255,255,0.1)", input_shadow="none", button_secondary_background_fill="#e8e8e8", button_secondary_background_fill_dark="#323232", button_secondary_text_color="#161616", button_secondary_text_color_dark="#fafafa", button_secondary_shadow="none", ) tokenizer = None model = None EOS_TOKEN_ID = None if not PREVIEW: tokenizer = AutoTokenizer.from_pretrained(MODEL_ID) model = AutoModelForCausalLM.from_pretrained( MODEL_ID, dtype=torch.bfloat16, attn_implementation="sdpa", ).to("cuda") model.eval() model.generation_config.use_cache = True EOS_TOKEN_ID = tokenizer.eos_token_id def _strip_thinking(text: str) -> str: if not isinstance(text, str): return "" return _THINK_RE.sub("", text).strip() def _render_math_symbols(text: str) -> str: """Keep common bare LaTeX operators readable outside math delimiters.""" for command, symbol in LATEX_SYMBOLS: text = re.sub(re.escape(command) + r"(?![A-Za-z])", symbol, text) return text # Math is typeset to MathML on the server. Gradio's client-side KaTeX swaps in # raw LaTeX on every streamed token and typesets it a frame later, so every # formula on screen flashed once per token. _CODE_RE = re.compile(r"(```[\s\S]*?(?:```|$)|`[^`\n]*`)") _MATH_RE = re.compile( r"\\\[(?P[\s\S]+?)\\\]" r"|\$\$(?P
[\s\S]+?)\$\$" r"|\\\((?P[\s\S]+?)\\\)" r"|(?[^\s$](?:[^$\n]*?[^\s$])?)\$(?!\w)" ) _MATH_OPENER_RE = re.compile(r"\\\[|\$\$|\\\(|(?]*\bdisplay="(block|inline)"[^>]*>.*?' r'(.*?).*?', re.DOTALL, ) # Characters that Markdown would act on inside the inline HTML. _MD_ESCAPES = str.maketrans({c: f"&#{ord(c)};" for c in "\\`*_[]~$|\n"}) _UNSAFE_LINK_RE = re.compile(r"\]\(\s*(?:javascript|vbscript|data):", re.IGNORECASE) _PARTIAL_BLOCK_RE = re.compile(r"\s*(?:[|>#=*+`$\\-]|\d+(?:[.)]|$))") _MATHML_TAGS = frozenset( "math semantics annotation mrow mi mn mo mtext mspace ms msup msub msubsup mfrac " "msqrt mroot mover munder munderover menclose mstyle mpadded mphantom mtable mtr mtd".split() ) _MATHML_ATTRS = frozenset( "xmlns display encoding accent accentunder columnalign columnlines columnspacing " "rowalign rowlines rowspacing displaystyle scriptlevel fence form largeop lspace " "rspace linebreak linethickness mathcolor mathbackground mathsize mathvariant " "maxsize minsize movablelimits notation separator stretchy symmetric width " "height depth voffset".split() ) _TAG_TOKEN_RE = re.compile(r"(<[^<>]*>)") _TAG_PARTS_RE = re.compile(r"<(/?)([a-zA-Z]+)\s*([^<>]*?)\s*(/?)>\Z") _ATTR_RE = re.compile(r'([a-zA-Z-]+)="([^"]*)"') _SAFE_ATTR_VALUE_RE = re.compile(r"[\w#.,%:/ ()+-]*\Z") def _sanitize_mathml(markup: str) -> str: # latex2mathml passes \text{...} through verbatim, so its output is # untrusted: keep only known MathML tags and inert attribute values. out = [] for token in _TAG_TOKEN_RE.split(markup): m = _TAG_PARTS_RE.match(token) if token.startswith("<") else None if m is None or m.group(2) not in _MATHML_TAGS: out.append(token.replace("<", "<")) continue closing, name, attrs, self_closing = m.groups() kept = "".join( f' {key}="{value}"' for key, value in _ATTR_RE.findall(attrs) if not closing and key in _MATHML_ATTRS and _SAFE_ATTR_VALUE_RE.match(value) ) out.append(f"<{closing}{name}{kept}{self_closing}>") return "".join(out) def _math_html(tex: str, display: bool) -> str | None: tex = tex.strip() try: body = latex_to_mathml(tex, display="block" if display else "inline") except Exception: return None open_end = body.index(">") + 1 inner = body[open_end : -len("")] note = html.escape(tex, quote=False) mathml = _sanitize_mathml( f"{body[:open_end]}{inner}" f'{note}' ) return mathml.translate(_MD_ESCAPES) def _restore_latex(text: str) -> str: """Turn displayed MathML back into the LaTeX the model originally wrote.""" def restore(m: re.Match) -> str: tex = html.unescape(m.group(2)) return f"\\[{tex}\\]" if m.group(1) == "block" else f"\\({tex}\\)" return _MATHML_RE.sub(restore, text) def _hold_back_partial(text: str) -> str: """Hide a still-streaming tail that would re-render as a different block.""" head, sep, tail = text.rpartition("\n") if tail and _PARTIAL_BLOCK_RE.match(tail) and tail.strip() not in ("\\]", "$$"): text = head + sep elif tail.count("**") % 2: text = head + sep + tail[: tail.rfind("**")] last = 0 for m in _MATH_RE.finditer(text): last = m.end() for m in _MATH_OPENER_RE.finditer(text, last): if m.group(0) == "$" and "\n" in text[m.end() :]: continue return text[: m.start()] if last < len(text) and (text.endswith("$") or re.search(r"(? str: # The chatbot renders with sanitize_html=False so MathML survives; escaping # "<" here is what keeps model and user text from injecting HTML. text = _render_math_symbols(text).replace("<", "<") return _UNSAFE_LINK_RE.sub("](#", text) def _display_markdown(text: str, streaming: bool = False) -> str: if streaming: text = _hold_back_partial(text) out = [] for i, part in enumerate(_CODE_RE.split(text)): if i % 2: out.append(part) continue last = 0 for m in _MATH_RE.finditer(part): out.append(_safe_text(part[last : m.start()])) tex = next(g for g in m.group("db", "dd", "ib", "id") if g is not None) display = m.group("db") is not None or m.group("dd") is not None out.append(_math_html(tex, display) or _safe_text(tex)) last = m.end() out.append(_safe_text(part[last:])) return "".join(out) def _parse_think(raw: str, assume_thinking: bool = False) -> tuple[str, str, bool]: """Return (think, answer, think_open) from a possibly streaming reply.""" open_idx = raw.find(THINK_OPEN) if open_idx == -1: # Some chat templates put the opening token in the generation # prompt. The streamer skips that prompt token, so output starts inside # the reasoning block and only emits . if assume_thinking: close_idx = raw.find(THINK_CLOSE) if close_idx != -1: return ( raw[:close_idx].strip(), raw[close_idx + len(THINK_CLOSE) :].strip(), False, ) visible = raw for i in range(len(THINK_CLOSE) - 1, 0, -1): if visible.endswith(THINK_CLOSE[:i]): visible = visible[:-i] break return visible.strip(), "", True # A partially streamed opening tag is not an answer yet. if raw.strip() and THINK_OPEN.startswith(raw.strip()): return "", "", False return "", raw, False rest = raw[open_idx + len(THINK_OPEN) :] close_idx = rest.find(THINK_CLOSE) before = raw[:open_idx] if close_idx == -1: # Hide a partially streamed closing tag at the end of the think block. for i in range(len(THINK_CLOSE) - 1, 0, -1): if rest.endswith(THINK_CLOSE[:i]): rest = rest[:-i] break return rest.strip(), before.strip(), True think = rest[:close_idx].strip() answer = (before + rest[close_idx + len(THINK_CLOSE) :]).strip() return think, answer, False def _gpu_seconds(history, max_new_tokens=2048, *args, **kwargs): # Scale the ZeroGPU reservation with the requested output budget. The # platform caps a single reservation at five minutes. tokens = int(max_new_tokens or 2048) return min(300, max(30, 20 + tokens // 50)) def _history_for_model(history: list) -> list[dict]: conversation = [{"role": "system", "content": SYSTEM_PROMPT}] for msg in history or []: role = msg.get("role", "user") meta = msg.get("metadata") or {} # Only skip thought/tool bubbles; Gradio attaches an empty metadata # dict to ordinary messages, so a bare truthiness check drops them all. is_thought = isinstance(meta, dict) and bool(meta.get("title")) if role == "system" or is_thought: continue content = msg.get("content", "") if isinstance(content, list): # Gradio 6 sends message content as a list of blocks. parts = [] for block in content: if isinstance(block, str): parts.append(block) elif isinstance(block, dict): text = block.get("text") or block.get("content") or "" if isinstance(text, str) and text: parts.append(text) content = "\n".join(parts) content = _restore_latex(content) if role == "assistant": content = _strip_thinking(content) if content: conversation.append({"role": role, "content": content}) return conversation def _stream_tokens(conversation: list[dict], max_new_tokens: int, temperature: float, top_p: float, enable_thinking: bool) -> Iterator[str]: if PREVIEW: import time demo = ( f"{THINK_OPEN}\n" "Check the premises, then the conclusion. Premise 1 gives Rain -> Wet. " "Premise 2 observes Wet. Inferring Rain from Wet affirms the consequent, " "which is invalid. The argument is a classic Barbara syllogism otherwise.\n" f"{THINK_CLOSE}\n\n" "Answer: entailment." ) acc = "" for ch in demo: acc += ch time.sleep(0.02) yield acc return encoded = tokenizer.apply_chat_template( conversation, add_generation_prompt=True, return_tensors="pt", return_dict=True, enable_thinking=enable_thinking, ) input_ids = encoded["input_ids"] attention_mask = encoded["attention_mask"] if input_ids.shape[1] > MAX_INPUT_TOKEN_LENGTH: input_ids = input_ids[:, -MAX_INPUT_TOKEN_LENGTH:] attention_mask = attention_mask[:, -MAX_INPUT_TOKEN_LENGTH:] gr.Warning(f"Trimmed the conversation to the last {MAX_INPUT_TOKEN_LENGTH} tokens.") input_ids = input_ids.to(model.device) attention_mask = attention_mask.to(model.device) streamer = TextIteratorStreamer( tokenizer, timeout=30.0, skip_prompt=True, skip_special_tokens=True ) generate_kwargs = dict( input_ids=input_ids, attention_mask=attention_mask, streamer=streamer, max_new_tokens=int(max_new_tokens), num_beams=1, use_cache=True, eos_token_id=EOS_TOKEN_ID, pad_token_id=tokenizer.pad_token_id or EOS_TOKEN_ID, ) if temperature == 0: generate_kwargs["do_sample"] = False else: generate_kwargs["do_sample"] = True generate_kwargs["temperature"] = float(temperature) generate_kwargs["top_p"] = float(top_p) Thread(target=model.generate, kwargs=generate_kwargs, daemon=True).start() acc = "" for text in streamer: acc += text yield acc def queue_message(message, history): """Paint the user turn immediately so the GPU wait is not a blank screen.""" message = (message or "").strip() if not message: return "", history, gr.skip() history = [dict(m) if isinstance(m, dict) else m for m in (history or [])] history.append({"role": "user", "content": _display_markdown(message)}) history.append( { "role": "assistant", "content": "", "metadata": {"title": "Logicizing…", "status": "pending"}, } ) return "", history, gr.update(visible=False) @spaces.GPU(duration=_gpu_seconds) def generate_reply( history, max_new_tokens, temperature, top_p, enable_thinking, show_thinking_trace, ): history = [dict(m) if isinstance(m, dict) else m for m in (history or [])] if not any(m.get("role") == "user" for m in history): yield history return conversation = _history_for_model(history) # Stream only into a bubble at the END of the history. Reaching for any # older assistant message would overwrite a previous turn in place. think_msg = None last = history[-1] if history else None if ( last is not None and last.get("role") == "assistant" and isinstance(last.get("metadata"), dict) and last["metadata"].get("status") == "pending" ): think_msg = last think_msg["content"] = "" elif last is not None and last.get("role") == "user": # Gradio strips the empty pending bubble in preprocessing; recreate it. think_msg = {"role": "assistant", "content": ""} history.append(think_msg) else: # No new user turn (e.g. an empty submit); don't regenerate old answers. yield history return think_msg["metadata"] = {"title": "Logicizing…", "status": "pending"} yield history answer_msg = {"role": "assistant", "content": ""} think, answer_text = "", "" def show_answer(text: str, streaming: bool) -> None: shown = _display_markdown(text, streaming=streaming) if not shown and not any(m is answer_msg for m in history): return if not show_thinking_trace and think_msg in history: history.remove(think_msg) answer_msg["content"] = shown if not any(m is answer_msg for m in history): history.append(answer_msg) for raw in _stream_tokens( conversation, max_new_tokens, temperature, top_p, enable_thinking ): think, answer, think_open = _parse_think( raw, assume_thinking=bool(enable_thinking) ) if show_thinking_trace and (think or think_open): think_msg["content"] = _display_markdown(think, streaming=think_open) think_msg["metadata"] = { "title": "Logicizing…" if think_open else "Logic", "status": "pending" if think_open else "done", } if answer: answer_text = answer show_answer(answer_text, streaming=True) elif not think_open and not think and raw.strip(): answer_text = raw.strip() show_answer(answer_text, streaming=True) elif any(m is answer_msg for m in history): # A partial tag was mistaken for an answer earlier; retract it. history.remove(answer_msg) answer_msg["content"] = "" answer_text = "" yield history if answer_text: show_answer(answer_text, streaming=False) if show_thinking_trace and think_msg["content"]: think_msg["content"] = _display_markdown(think) think_msg["metadata"] = {"title": "Logic", "status": "done"} yield history elif think_msg in history: history.remove(think_msg) yield history with gr.Blocks(fill_height=True, fill_width=True, elem_id="app-root") as demo: with gr.Column(elem_id="app-shell"): gr.HTML(HEADER) with gr.Tabs(elem_id="twil-tabs") as tabs: with gr.Tab("Chat", id="chat", elem_id="tab-chat"): chatbot = gr.Chatbot( value=[], placeholder=PLACEHOLDER, label="", show_label=False, layout="bubble", avatar_images=(None, str(AVATAR) if AVATAR.exists() else None), allow_tags=["think"], sanitize_html=False, reasoning_tags=[(THINK_OPEN, THINK_CLOSE)], latex_delimiters=[], buttons=None, editable=False, line_breaks=False, elem_id="twil-chat", scale=1, height="100%", autoscroll=False, ) with gr.Row(elem_id="chips-row") as chips_row: chip_btns = [ gr.Button(label, variant="secondary", size="sm", elem_classes=["twil-chip"]) for label, _ in CHIPS ] with gr.Row(elem_id="composer-row"): prompt = gr.Textbox( placeholder="Ask webAI anything…", show_label=False, container=False, lines=2, max_lines=2, elem_id="twil-input", scale=8, ) send = gr.Button("↑", elem_id="send-btn", scale=0) with gr.Accordion("Parameters", open=False, elem_id="params-box"): max_new_tokens = gr.Slider(256, 16384, value=2048, step=256, label="Max new tokens") temperature = gr.Slider(0, 1.5, value=0, step=0.05, label="Temperature (0 = greedy)") top_p = gr.Slider(0.1, 1.0, value=0.95, step=0.05, label="Top-p") enable_thinking = gr.Checkbox(value=True, label="Enable thinking") show_thinking_trace = gr.Checkbox( value=True, label="Show thinking trace (use its arrow to expand or collapse)", ) gr.HTML('

This is AI and it can make mistakes

') with gr.Tab("Showcase", id="showcase", elem_id="tab-showcase"): gr.HTML(SHOWCASE_INTRO) card_btns = [] for start in range(0, len(SHOWCASE), 3): with gr.Row(elem_classes=["twil-card-row"]): for case in SHOWCASE[start : start + 3]: with gr.Column(elem_classes=["twil-card"], scale=1, min_width=220): gr.HTML(_card_html(case)) card_btns.append( gr.Button("Analyze it", size="sm", elem_classes=["twil-card-btn"]) ) # Lock the composer while a turn is running: overlapping events make the # finishing stream overwrite the chat with its own stale history. lockable = [prompt, send, *card_btns] def _lock(): return [gr.update(interactive=False)] * len(lockable) def _unlock(): return [gr.update(interactive=True)] * len(lockable) def _wire(event, prep_fn=None, prep_outputs=None): chain = event(_lock, None, lockable, show_progress="hidden") if prep_fn is not None: chain = chain.then(prep_fn, outputs=prep_outputs, show_progress="hidden") chain = chain.then( queue_message, [prompt, chatbot], [prompt, chatbot, chips_row], show_progress="hidden", ).then( generate_reply, [ chatbot, max_new_tokens, temperature, top_p, enable_thinking, show_thinking_trace, ], chatbot, concurrency_limit=1, show_progress="hidden", ) chain.then(_unlock, None, lockable, show_progress="hidden") _wire(send.click) _wire(prompt.submit) for btn, (_, template) in zip(chip_btns, CHIPS): btn.click(lambda t=template: t, outputs=prompt) for btn, case in zip(card_btns, SHOWCASE): case_prompt = SHOWCASE_TEMPLATE.format(argument=case["quote"]) _wire( btn.click, prep_fn=lambda p=case_prompt: (p, gr.Tabs(selected="chat")), prep_outputs=[prompt, tabs], ) if __name__ == "__main__": demo.launch( theme=THEME, css=CSS, head=HEAD, allowed_paths=[str(ASSETS), str(ASSETS.parent)], ssr_mode=False, show_error=True, )