ReadingCoach / app.py
NLPV's picture
Upload 7 files
4eab58f verified
Raw
History Blame Contribute Delete
5.55 kB
"""
Hugging Face Space entrypoint: Hindi / English reading & pronunciation coach.
Set ASR_BACKEND (and GROQ_API_KEY as a Space secret) in Settings -> Variables.
"""
from __future__ import annotations
import hashlib
import os
import tempfile
import gradio as gr
import pandas as pd
from asr_backends import get_backend
from scoring import score
LANGS = {"हिंदी (Hindi)": "hi", "English": "en"}
SAMPLES = {
"hi": "सूरज सुबह जल्दी उठता है और अपनी किताब पढ़ता है। उसे कहानियाँ बहुत पसंद हैं।",
"en": "The quick brown fox jumps over the lazy dog while the children watch quietly.",
}
# ---------------------------------------------------------------------------
# TTS -- cached, because on a Space every gTTS call is a network round trip
# ---------------------------------------------------------------------------
TTS_DIR = os.path.join(tempfile.gettempdir(), "coach_tts")
os.makedirs(TTS_DIR, exist_ok=True)
def speak(text: str, lang_label: str, slow: bool):
if not (text or "").strip():
raise gr.Error("Type or paste a passage first.")
lang = LANGS[lang_label]
key = hashlib.sha1(f"{lang}|{slow}|{text}".encode()).hexdigest()[:20]
path = os.path.join(TTS_DIR, f"{key}.mp3")
if not os.path.exists(path):
from gtts import gTTS
try:
gTTS(text=text, lang=lang, slow=slow).save(path)
except Exception as exc:
raise gr.Error(f"Text-to-speech unavailable: {exc}") from exc
return path
def fill_sample(lang_label: str) -> str:
return SAMPLES[LANGS[lang_label]]
# ---------------------------------------------------------------------------
# Evaluation
# ---------------------------------------------------------------------------
EMPTY = pd.DataFrame(columns=["अपेक्षित / Expected", "सुना गया / Heard",
"प्रकार / Error type", "समानता / Similarity"])
def evaluate(audio_path, passage, lang_label, lenient_decoding):
if not audio_path:
return {"error": "No recording received — record or upload audio first."}, EMPTY
if not (passage or "").strip():
return {"error": "Paste the passage to read first."}, EMPTY
lang = LANGS[lang_label]
try:
backend = get_backend()
hint = passage.strip() if lenient_decoding else None
tr = backend.transcribe(audio_path, lang, hint=hint)
except Exception as exc:
return {"error": f"{type(exc).__name__}: {exc}"}, EMPTY
if not tr.text:
return {"error": "Nothing was transcribed. Check the mic level and try again."}, EMPTY
return score(passage, tr, lang)
# ---------------------------------------------------------------------------
# UI
# ---------------------------------------------------------------------------
def build() -> gr.Blocks:
with gr.Blocks(title="Reading & Pronunciation Coach") as app:
gr.Markdown("## 🗣️ Reading & Pronunciation Coach — हिंदी / English")
status = gr.Markdown("Resolving ASR backend…")
with gr.Row():
lang = gr.Dropdown(list(LANGS), value="हिंदी (Hindi)", label="Language", scale=2)
sample_btn = gr.Button("Load sample passage", scale=1)
passage = gr.Textbox(
label="Passage to read", lines=4,
placeholder="यहाँ हिंदी टेक्स्ट लिखें… / Paste English text here…",
)
with gr.Row():
slow = gr.Checkbox(label="Slow speech", value=False, scale=1)
listen = gr.Button("🔊 Listen", scale=1)
tts_audio = gr.Audio(label="Model reading", type="filepath")
gr.Markdown("### 🎤 Now read it aloud")
mic = gr.Audio(sources=["microphone", "upload"], type="filepath",
label="Your recording")
lenient = gr.Checkbox(
label="Lenient decoding — biases the ASR toward the passage. "
"Scores look better but real mistakes get hidden. Leave off for assessment.",
value=False,
)
submit = gr.Button("✅ Check my reading", variant="primary")
metrics = gr.JSON(label="Results")
table = gr.Dataframe(label="गलती तालिका / Error table", wrap=True)
gr.Markdown(
"**Reading the scores.** *Word accuracy* is strict: a word counts only if it "
"matches exactly. *Lenient score* gives partial credit by similarity, so a "
"near-miss on a hard word is not treated like a skipped line. *WER* is the "
"standard ASR metric and includes extra words, so it can exceed the accuracy gap."
)
def backend_line():
try:
b = get_backend()
extra = "" if b.supports_word_confidence else \
" · per-word confidence unavailable on this backend"
return f"**Backend:** {b.describe()}{extra}"
except Exception as exc:
return f"⚠️ **Backend not ready:** {exc}"
app.load(backend_line, None, status)
sample_btn.click(fill_sample, lang, passage)
listen.click(speak, [passage, lang, slow], tts_audio)
submit.click(evaluate, [mic, passage, lang, lenient], [metrics, table])
return app
if __name__ == "__main__":
build().queue(max_size=12).launch()