Translator / app.py
qadeer1008's picture
Upload 3 files
7420de0 verified
Raw
History Blame Contribute Delete
8.78 kB
# ============================================================================
# ENGLISH → URDU AI TRANSLATOR — app.py (Hugging Face Spaces ready)
# Deploy: create a new Space (SDK: Gradio), upload this file + requirements.txt
# ============================================================================
# ----------------------------------------------------------------------------
# 1. IMPORTS
# ----------------------------------------------------------------------------
import re
import torch
import gradio as gr
from transformers import MarianMTModel, MarianTokenizer
# ----------------------------------------------------------------------------
# 2. LOAD THE FREE ENGLISH → URDU MODEL
# ----------------------------------------------------------------------------
# Helsinki-NLP/opus-mt-en-ur is a free, open-source MarianMT model trained
# specifically for English -> Urdu translation. It downloads automatically
# from the Hugging Face Hub on first run and is cached afterwards.
MODEL_NAME = "Helsinki-NLP/opus-mt-en-ur"
print("Loading translation model...")
device = "cuda" if torch.cuda.is_available() else "cpu"
try:
tokenizer = MarianTokenizer.from_pretrained(MODEL_NAME)
model = MarianMTModel.from_pretrained(MODEL_NAME).to(device)
model.eval()
print(f"Model loaded successfully on device: {device}")
except Exception as e:
raise RuntimeError(f"Failed to load the translation model: {e}")
# ----------------------------------------------------------------------------
# 3. TRANSLATION LOGIC
# ----------------------------------------------------------------------------
def split_into_sentences(text: str):
"""
Split a paragraph into sentences so long, multi-sentence text is
translated more accurately (MarianMT works best on shorter chunks).
Keeps punctuation attached to each sentence.
"""
sentences = re.split(r'(?<=[.!?])\s+', text.strip())
return [s for s in sentences if s.strip()]
def translate_en_to_ur(text: str) -> str:
"""
Translates English text (single sentence, multiple sentences, or a
full paragraph) into Urdu using the loaded MarianMT model.
Punctuation and paragraph breaks are preserved.
"""
if not text or not text.strip():
return ""
# Preserve paragraph breaks by translating each paragraph separately
paragraphs = text.split("\n")
translated_paragraphs = []
for paragraph in paragraphs:
if not paragraph.strip():
translated_paragraphs.append("")
continue
sentences = split_into_sentences(paragraph)
translated_sentences = []
for sentence in sentences:
inputs = tokenizer(sentence, return_tensors="pt", padding=True, truncation=True).to(device)
with torch.no_grad():
translated_tokens = model.generate(**inputs, max_length=512, num_beams=4)
translated_text = tokenizer.decode(translated_tokens[0], skip_special_tokens=True)
translated_sentences.append(translated_text)
translated_paragraphs.append(" ".join(translated_sentences))
return "\n".join(translated_paragraphs)
def handle_translate(english_text: str, progress=gr.Progress()):
"""
Wrapper function called by the Gradio 'Translate' button.
Handles empty input, shows progress, and catches errors gracefully.
"""
if not english_text or not english_text.strip():
return "⚠️ Please enter some English text to translate."
try:
progress(0.2, desc="Analyzing text...")
progress(0.5, desc="Translating to Urdu...")
result = translate_en_to_ur(english_text)
progress(1.0, desc="Done!")
if not result.strip():
return "⚠️ Could not generate a translation. Please try different text."
return result
except Exception as e:
return f"❌ An error occurred during translation: {str(e)}"
def count_characters(text: str) -> str:
"""Returns a live character/word count string for the input box."""
if not text:
return "0 characters | 0 words"
char_count = len(text)
word_count = len(text.split())
return f"{char_count} characters | {word_count} words"
def clear_fields():
"""Resets the input box, output box, and character counter."""
return "", "", "0 characters | 0 words"
# ----------------------------------------------------------------------------
# 4. CUSTOM CSS — MODERN, ROUNDED, SOFT-COLORED, RESPONSIVE UI
# ----------------------------------------------------------------------------
custom_css = """
.gradio-container {
font-family: 'Segoe UI', 'Poppins', sans-serif !important;
background: linear-gradient(135deg, #f5f7fa 0%, #e8eef7 100%) !important;
max-width: 900px !important;
margin: auto !important;
}
#title_md h1 {
text-align: center;
font-weight: 700;
background: linear-gradient(90deg, #2b6cb0, #38b2ac);
-webkit-background-clip: text;
-webkit-text-fill-color: transparent;
margin-bottom: 0px;
}
#subtitle_md {
text-align: center;
color: #555;
margin-top: 4px;
margin-bottom: 20px;
}
.gr-box, .block {
border-radius: 16px !important;
box-shadow: 0 4px 12px rgba(0,0,0,0.06) !important;
}
textarea, input {
border-radius: 14px !important;
border: 1px solid #d6e0ea !important;
}
#translate_btn {
background: linear-gradient(90deg, #2b6cb0, #38b2ac) !important;
color: white !important;
border-radius: 14px !important;
font-weight: 600 !important;
border: none !important;
}
#clear_btn {
border-radius: 14px !important;
font-weight: 600 !important;
background: #f1f3f6 !important;
color: #333 !important;
border: 1px solid #ddd !important;
}
#char_counter {
text-align: right;
color: #888;
font-size: 0.85em;
}
footer {
visibility: hidden;
}
"""
# ----------------------------------------------------------------------------
# 5. EXAMPLE SENTENCES
# ----------------------------------------------------------------------------
example_sentences = [
"Hello, how are you today?",
"I love reading books in my free time.",
"The weather is beautiful this morning.",
"Can you please help me with my homework?",
"Pakistan is a country with a rich cultural heritage.",
"Artificial intelligence is changing the world rapidly.",
"Thank you very much for your kindness.",
"Education is the key to a better future.",
]
# ----------------------------------------------------------------------------
# 6. BUILD THE GRADIO BLOCKS INTERFACE
# ----------------------------------------------------------------------------
with gr.Blocks(css=custom_css, theme=gr.themes.Soft(primary_hue="blue", secondary_hue="teal")) as demo:
gr.Markdown("# 🌐 English → Urdu AI Translator", elem_id="title_md")
gr.Markdown(
"Translate English text into fluent Urdu instantly, powered by a free open-source "
"Hugging Face model. Supports single sentences, multiple sentences, and full paragraphs.",
elem_id="subtitle_md"
)
with gr.Row():
with gr.Column(scale=1):
english_input = gr.Textbox(
label="✍️ English Text",
placeholder="Type or paste English text here...",
lines=8,
)
char_counter = gr.Markdown("0 characters | 0 words", elem_id="char_counter")
with gr.Row():
translate_btn = gr.Button("🔁 Translate", elem_id="translate_btn", variant="primary")
clear_btn = gr.Button("🗑️ Clear", elem_id="clear_btn")
with gr.Column(scale=1):
urdu_output = gr.Textbox(
label="🇵🇰 Urdu Translation",
placeholder="اردو ترجمہ یہاں ظاہر ہوگا...",
lines=8,
rtl=True,
interactive=False,
)
gr.Markdown("### 💡 Try an example:")
gr.Examples(
examples=example_sentences,
inputs=english_input,
label="Example Sentences",
)
# Event wiring
english_input.change(fn=count_characters, inputs=english_input, outputs=char_counter)
translate_btn.click(fn=handle_translate, inputs=english_input, outputs=urdu_output)
english_input.submit(fn=handle_translate, inputs=english_input, outputs=urdu_output)
clear_btn.click(fn=clear_fields, inputs=None, outputs=[english_input, urdu_output, char_counter])
# ----------------------------------------------------------------------------
# 7. LAUNCH — works both locally and on Hugging Face Spaces
# ----------------------------------------------------------------------------
if __name__ == "__main__":
demo.launch()