Spaces:
Runtime error
Runtime error
File size: 6,210 Bytes
6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d 44243d1 6f91a2d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 | import re
from collections import Counter
from typing import List, Tuple
import gradio as gr
# --- NLTK setup ---
import nltk
from nltk.tokenize import word_tokenize, sent_tokenize
STOPWORDS = set()
PUNCT_RE = re.compile(r"[^\w\s]") # remove punctuation (keeps letters, digits, underscore, and spaces)
def ensure_nltk():
"""Download required NLTK resources and load stopwords."""
global STOPWORDS
try:
nltk.download("punkt")
nltk.download("stopwords")
from nltk.corpus import stopwords
STOPWORDS = set(stopwords.words("english"))
return True, "NLTK data installed (punkt, stopwords)."
except Exception as e:
return False, f"NLTK install error: {e}"
def read_text_area_or_file(f, text):
"""Priority: uploaded file if present, else textbox. Supports .txt and .docx"""
if f is not None:
path = getattr(f, "name", None)
if path and path.lower().endswith(".txt"):
with open(path, "r", encoding="utf-8", errors="ignore") as fh:
return fh.read()
elif path and path.lower().endswith(".docx"):
import docx2txt
return docx2txt.process(path) or ""
else:
return f"Unsupported file type. Please upload .txt or .docx."
return text or ""
def normalize_and_tokenize(text: str) -> Tuple[List[str], List[List[str]]]:
"""Lowercase, remove punctuation, drop stopwords, return sentences + tokenized sentences"""
lowered = text.lower()
no_punct = PUNCT_RE.sub(" ", lowered)
sentences = [s.strip() for s in sent_tokenize(no_punct) if s.strip()]
if not sentences:
rough = re.split(r"[.\n]+", no_punct)
sentences = [s.strip() for s in rough if s.strip()]
tokenized = []
for s in sentences:
toks = [t for t in word_tokenize(s) if t and t not in STOPWORDS and not t.isnumeric()]
tokenized.append(toks)
return sentences, tokenized
def build_bow(tokenized_sentences: List[List[str]]) -> Counter:
all_words = [w for sent in tokenized_sentences for w in sent]
return Counter(all_words)
def vectorize_sentence(tokenized_sentences: List[List[str]], sentence_index: int, vocabulary: List[str]) -> List[int]:
if not tokenized_sentences:
return []
idx = max(0, min(sentence_index, len(tokenized_sentences) - 1))
sent_tokens = tokenized_sentences[idx]
return [sent_tokens.count(word) for word in vocabulary]
def process(menu_choice, text, file, sentence_index):
if menu_choice == "Install NLTK":
ok, msg = ensure_nltk()
return gr.update(value=msg), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
raw = read_text_area_or_file(file, text)
if not raw or raw.startswith("Unsupported file type"):
return gr.update(value=raw or "No input text found."), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
sentences, tokenized = normalize_and_tokenize(raw)
if menu_choice == "Tokenize sentences into words":
lines = [f"Sentence {i+1}: {toks}" for i, toks in enumerate(tokenized)]
return gr.update(value="\n".join(lines)), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
bow = build_bow(tokenized)
if menu_choice == "Count word occurrences (Bag of Words)":
items = sorted(bow.items(), key=lambda x: (-x[1], x[0]))
text_out = "Bag of Words (word -> count):\n" + "\n".join([f"{w}: {c}" for w, c in items])
return gr.update(value=text_out), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
if menu_choice == "Build a word frequency vector for any selected sentence":
vocab = sorted(bow.keys())
vec = vectorize_sentence(tokenized, sentence_index, vocab)
chosen_idx = max(0, min(sentence_index, len(tokenized) - 1))
chosen_sentence = " ".join(sentences[chosen_idx].split())
text_summary = (
f"Selected sentence index: {chosen_idx} (1-based: {chosen_idx+1})\n"
f"Sentence tokens: {tokenized[chosen_idx]}\n"
f"Vocabulary size: {len(vocab)}\n"
f"Vector length: {len(vec)}"
)
vocab_text = ", ".join(vocab) if vocab else "(empty vocabulary)"
vector_text = ", ".join(map(str, vec)) if vec else "(empty vector)"
return (
gr.update(value=text_summary + f"\n\nSentence (cleaned): {chosen_sentence}"),
gr.update(value=vocab_text, visible=True, label="Vocabulary (ordered)"),
gr.update(value=vector_text, visible=True, label="Word Frequency Vector (aligned with vocabulary)"),
gr.update(visible=True)
)
return gr.update(value="Choose a menu option."), gr.update(visible=False), gr.update(visible=False), gr.update(visible=False)
with gr.Blocks(title="BoW + Vectorizer (NLTK + Gradio)") as demo:
gr.Markdown("# NLP Utilities: Tokenize • Bag of Words • Sentence Vector\nText is lowercased, punctuation removed, and stopwords dropped.")
menu = gr.Radio(
choices=[
"Install NLTK",
"Tokenize sentences into words",
"Count word occurrences (Bag of Words)",
"Build a word frequency vector for any selected sentence",
],
value="Tokenize sentences into words",
label="Menu"
)
text_in = gr.Textbox(lines=8, label="Text", placeholder="Paste text here...")
file_in = gr.File(label="Or drop a .txt / .docx file", file_types=[".txt", ".docx"])
sentence_index = gr.Number(value=0, precision=0, visible=False, label="Sentence index (0-based)")
run_btn = gr.Button("Process", variant="primary")
out_main = gr.Textbox(label="Output", lines=12)
out_vocab = gr.Textbox(label="Vocabulary (ordered)", lines=6, visible=False)
out_vector = gr.Textbox(label="Word Frequency Vector (aligned with vocabulary)", lines=6, visible=False)
run_btn.click(fn=process, inputs=[menu, text_in, file_in, sentence_index], outputs=[out_main, out_vocab, out_vector, sentence_index])
if __name__ == "__main__":
# safe: pre-install NLTK resources at startup
ensure_nltk()
demo.launch()
|