LabDemo2 / app.py
eaglelandsonce's picture
Update app.py
44243d1 verified
Raw
History Blame Contribute Delete
6.21 kB
import re
from collections import Counter
from typing import List, Tuple
import gradio as gr
# --- NLTK setup ---
import nltk
from nltk.tokenize import word_tokenize, sent_tokenize
STOPWORDS = set()
PUNCT_RE = re.compile(r"[^\w\s]") # remove punctuation (keeps letters, digits, underscore, and spaces)
def ensure_nltk():
"""Download required NLTK resources and load stopwords."""
global STOPWORDS
try:
nltk.download("punkt")
nltk.download("stopwords")
from nltk.corpus import stopwords
STOPWORDS = set(stopwords.words("english"))
return True, "NLTK data installed (punkt, stopwords)."
except Exception as e:
return False, f"NLTK install error: {e}"
def read_text_area_or_file(f, text):
"""Priority: uploaded file if present, else textbox. Supports .txt and .docx"""
if f is not None:
path = getattr(f, "name", None)
if path and path.lower().endswith(".txt"):
with open(path, "r", encoding="utf-8", errors="ignore") as fh:
return fh.read()
elif path and path.lower().endswith(".docx"):
import docx2txt
return docx2txt.process(path) or ""
else:
return f"Unsupported file type. Please upload .txt or .docx."
return text or ""
def normalize_and_tokenize(text: str) -> Tuple[List[str], List[List[str]]]:
"""Lowercase, remove punctuation, drop stopwords, return sentences + tokenized sentences"""
lowered = text.lower()
no_punct = PUNCT_RE.sub(" ", lowered)
sentences = [s.strip() for s in sent_tokenize(no_punct) if s.strip()]
if not sentences:
rough = re.split(r"[.\n]+", no_punct)
sentences = [s.strip() for s in rough if s.strip()]
tokenized = []
for s in sentences:
toks = [t for t in word_tokenize(s) if t and t not in STOPWORDS and not t.isnumeric()]
tokenized.append(toks)
return sentences, tokenized
def build_bow(tokenized_sentences: List[List[str]]) -> Counter:
all_words = [w for sent in tokenized_sentences for w in sent]
return Counter(all_words)
def vectorize_sentence(tokenized_sentences: List[List[str]], sentence_index: int, vocabulary: List[str]) -> List[int]:
if not tokenized_sentences:
return []
idx = max(0, min(sentence_index, len(tokenized_sentences) - 1))
sent_tokens = tokenized_sentences[idx]
return [sent_tokens.count(word) for word in vocabulary]
def process(menu_choice, text, file, sentence_index):
if menu_choice == "Install NLTK":
ok, msg = ensure_nltk()
return gr.update(value=msg), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
raw = read_text_area_or_file(file, text)
if not raw or raw.startswith("Unsupported file type"):
return gr.update(value=raw or "No input text found."), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
sentences, tokenized = normalize_and_tokenize(raw)
if menu_choice == "Tokenize sentences into words":
lines = [f"Sentence {i+1}: {toks}" for i, toks in enumerate(tokenized)]
return gr.update(value="\n".join(lines)), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
bow = build_bow(tokenized)
if menu_choice == "Count word occurrences (Bag of Words)":
items = sorted(bow.items(), key=lambda x: (-x[1], x[0]))
text_out = "Bag of Words (word -> count):\n" + "\n".join([f"{w}: {c}" for w, c in items])
return gr.update(value=text_out), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
if menu_choice == "Build a word frequency vector for any selected sentence":
vocab = sorted(bow.keys())
vec = vectorize_sentence(tokenized, sentence_index, vocab)
chosen_idx = max(0, min(sentence_index, len(tokenized) - 1))
chosen_sentence = " ".join(sentences[chosen_idx].split())
text_summary = (
f"Selected sentence index: {chosen_idx} (1-based: {chosen_idx+1})\n"
f"Sentence tokens: {tokenized[chosen_idx]}\n"
f"Vocabulary size: {len(vocab)}\n"
f"Vector length: {len(vec)}"
)
vocab_text = ", ".join(vocab) if vocab else "(empty vocabulary)"
vector_text = ", ".join(map(str, vec)) if vec else "(empty vector)"
return (
gr.update(value=text_summary + f"\n\nSentence (cleaned): {chosen_sentence}"),
gr.update(value=vocab_text, visible=True, label="Vocabulary (ordered)"),
gr.update(value=vector_text, visible=True, label="Word Frequency Vector (aligned with vocabulary)"),
gr.update(visible=True)
)
return gr.update(value="Choose a menu option."), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
with gr.Blocks(title="BoW + Vectorizer (NLTK + Gradio)") as demo:
gr.Markdown("# NLP Utilities: Tokenize • Bag of Words • Sentence Vector\nText is lowercased, punctuation removed, and stopwords dropped.")
menu = gr.Radio(
choices=[
"Install NLTK",
"Tokenize sentences into words",
"Count word occurrences (Bag of Words)",
"Build a word frequency vector for any selected sentence",
],
value="Tokenize sentences into words",
label="Menu"
)
text_in = gr.Textbox(lines=8, label="Text", placeholder="Paste text here...")
file_in = gr.File(label="Or drop a .txt / .docx file", file_types=[".txt", ".docx"])
sentence_index = gr.Number(value=0, precision=0, visible=False, label="Sentence index (0-based)")
run_btn = gr.Button("Process", variant="primary")
out_main = gr.Textbox(label="Output", lines=12)
out_vocab = gr.Textbox(label="Vocabulary (ordered)", lines=6, visible=False)
out_vector = gr.Textbox(label="Word Frequency Vector (aligned with vocabulary)", lines=6, visible=False)
run_btn.click(fn=process, inputs=[menu, text_in, file_in, sentence_index], outputs=[out_main, out_vocab, out_vector, sentence_index])
if __name__ == "__main__":
# safe: pre-install NLTK resources at startup
ensure_nltk()
demo.launch()