"""OpenTextShield demo Space. Loads the OpenTextShield mBERT model from the Hub and classifies an SMS as ham (legitimate), spam or phishing, applying the same text normalisation as the production API so obfuscated messages are handled identically. Styled to match the OpenTextShield / TelecomsXChange (TCXC) brand. """ import gradio as gr import torch from transformers import AutoModelForSequenceClassification, AutoTokenizer from normalizer import normalize_unicode MODEL_ID = "telecomsxchange/OpenTextShield" MAX_TOKENS = 96 # matches the production API's truncation length tokenizer = AutoTokenizer.from_pretrained(MODEL_ID) model = AutoModelForSequenceClassification.from_pretrained(MODEL_ID) model.eval() DISPLAY = { "ham": ("Legitimate", "ham", "reads like a normal message"), "spam": ("Spam", "spam", "unwanted promotional or bulk content"), "phishing": ("Phishing", "phishing", "an attempt to steal credentials, money or personal data"), } EXAMPLES = [ "Running about 15 min late, order me the usual? I'll grab the bill.", "USPS: Your parcel could not be delivered because of an unpaid customs fee. Settle it within 24h to avoid return: http://usps-redelivery.top/pay", "BBVA: Hemos detectado un acceso inusual a su cuenta. Verifique su identidad ahora para evitar el bloqueo: http://bbva-seguridad.info/verificar", "CONGRATULATIONS! Your number was picked for a $1,000 gift card. Reply YES to claim before midnight!", "Paypal: unusual sign-in detected. Confirm your identity: http://рayрal-id.com/verify", ] def classify(message: str): message = (message or "").strip() if not message: return "", {}, "" normalized = normalize_unicode(message) inputs = tokenizer( normalized, return_tensors="pt", truncation=True, max_length=MAX_TOKENS, ) with torch.inference_mode(): probs = torch.softmax(model(**inputs).logits, dim=-1)[0] top = int(probs.argmax()) raw = model.config.id2label[top] word, css_class, meaning = DISPLAY[raw] verdict_html = ( f'
{word}' f'{probs[top]:.1%}' f'— {meaning}
' ) scores = {DISPLAY[model.config.id2label[i]][0]: float(p) for i, p in enumerate(probs)} note = ( "" if normalized == message else f"**Obfuscation detected** — classified as the text it imitates:\n\n> {normalized}" ) return verdict_html, scores, note MARK_SVG = """""" HEADER = f"""
{MARK_SVG}
OpenTextShield
Open-source SMS spam & phishing detection, in many languages
""" ARTICLE = """ ### About OpenTextShield is an open-source model that detects SMS spam and phishing (smishing) in many languages. It is used by telecom carriers to screen live SMS traffic and protect subscribers in real networks, and it runs entirely on your own servers — as a REST API, an SMPP proxy in front of your SMSC, or both. No third-party AI service is involved. This demo applies the same Unicode normalisation as the production API, so zero-width, full-width, homoglyph and leetspeak disguises are classified as the text they imitate. ### Run it yourself ```bash docker pull telecomsxchange/opentextshield:latest docker run -d -p 8002:8002 -p 8080:8080 telecomsxchange/opentextshield:latest curl -X POST "http://localhost:8002/predict/" \\ -H "Content-Type: application/json" \\ -d '{"text":"Your account has been suspended. Verify now at http://secure-login-check.xyz","model":"ots-mbert"}' ``` Or load the model directly: ```python from transformers import pipeline classifier = pipeline("text-classification", model="telecomsxchange/OpenTextShield") classifier("Your package is held. Pay the fee: http://usps-redelivery.top/pay") ``` Built by [TelecomsXChange (TCXC)](https://www.telecomsxchange.com) · MIT licensed """ CSS = """ :root { --ots-paper:#F2F3F1; --ots-surface:#FAFAF9; --ots-ink:#16181A; --ots-muted:#6B7076; --ots-hair:#D9DCD8; --ots-ham:#1E6E50; --ots-spam:#A6621A; --ots-phishing:#B42318; } .dark { --ots-paper:#15171A; --ots-surface:#1C1F23; --ots-ink:#E9EBEC; --ots-muted:#9AA0A6; --ots-hair:#2C3035; --ots-ham:#4CC38A; --ots-spam:#E5A04D; --ots-phishing:#F0655A; } .gradio-container { max-width: min(780px, 100%) !important; margin: 0 auto !important; } .ots-header { display:flex; align-items:flex-end; justify-content:space-between; flex-wrap:wrap; gap:12px; padding:8px 0 4px; border-bottom:1px solid var(--ots-hair); } .ots-brand { display:flex; align-items:center; gap:14px; } .ots-mark { width:40px; height:45px; flex:none; } .ots-mark-shield { fill: var(--ots-ink); } .ots-mark-check { stroke: var(--ots-paper); } .ots-title { font-size:22px; font-weight:700; letter-spacing:-0.02em; color:var(--ots-ink); } .ots-sub { font-size:13.5px; color:var(--ots-muted); } .ots-links { display:flex; gap:14px; font-size:13.5px; padding-bottom:4px; } .ots-links a { color:var(--ots-muted) !important; text-decoration:none !important; border-bottom:1px solid var(--ots-hair); } .ots-links a:hover { color:var(--ots-ink) !important; border-color:var(--ots-ink); } .ots-verdict { display:flex; align-items:baseline; gap:10px; flex-wrap:wrap; padding:6px 2px 2px; min-height:40px; } .ots-word { font-size:30px; font-weight:700; letter-spacing:-0.02em; line-height:1.1; } .ots-ham { color:var(--ots-ham); } .ots-spam { color:var(--ots-spam); } .ots-phishing { color:var(--ots-phishing); } .ots-conf { font-size:17px; color:var(--ots-ink); font-variant-numeric:tabular-nums; } .ots-note { font-size:13.5px; color:var(--ots-muted); } @media (max-width: 640px) { .gradio-container { padding-left:16px !important; padding-right:16px !important; } .ots-header { flex-direction:column; align-items:flex-start; gap:8px; } .ots-brand { gap:10px; } .ots-mark { width:32px; height:36px; } .ots-title { font-size:19px; } .ots-sub { font-size:12.5px; } .ots-links { flex-wrap:wrap; gap:10px; } .ots-word { font-size:24px; } .ots-conf { font-size:15px; } } """ theme = gr.themes.Soft( primary_hue=gr.themes.colors.stone, neutral_hue=gr.themes.colors.stone, font=[gr.themes.GoogleFont("Schibsted Grotesk"), "system-ui", "sans-serif"], font_mono=[gr.themes.GoogleFont("JetBrains Mono"), "ui-monospace", "monospace"], ).set( body_background_fill="#F2F3F1", body_background_fill_dark="#15171A", block_background_fill="#FAFAF9", block_background_fill_dark="#1C1F23", button_primary_background_fill="#16181A", button_primary_background_fill_hover="#2C3035", button_primary_text_color="#FAFAF9", button_primary_background_fill_dark="#E9EBEC", button_primary_background_fill_hover_dark="#FFFFFF", button_primary_text_color_dark="#15171A", block_title_text_color="#6B7076", block_title_text_color_dark="#9AA0A6", ) with gr.Blocks(theme=theme, css=CSS, title="OpenTextShield — SMS Spam & Phishing Detection") as demo: gr.HTML(HEADER) msg = gr.Textbox( lines=3, max_lines=8, label="Text message", placeholder="Paste an SMS to check…", ) check = gr.Button("Check message", variant="primary") verdict = gr.HTML(label="Verdict") scores = gr.Label(num_top_classes=3, label="Confidence", show_label=True) note = gr.Markdown() gr.Examples( examples=[[e] for e in EXAMPLES], inputs=msg, outputs=[verdict, scores, note], fn=classify, run_on_click=True, cache_examples=False, label="Try a sample", ) gr.Markdown(ARTICLE) check.click(classify, inputs=msg, outputs=[verdict, scores, note]) msg.submit(classify, inputs=msg, outputs=[verdict, scores, note]) if __name__ == "__main__": demo.launch()