Sentiment_Analysis / modules /preprocessing.py
Pro-Coder's picture
Create preprocessing.py
c7e6faa verified
Raw
History Blame Contribute Delete
935 Bytes
# modules/preprocessing.py
import re
import nltk
# Download stopwords if not already present
nltk.download("stopwords", quiet=True)
from nltk.corpus import stopwords
STOPWORDS = set(stopwords.words("english"))
def clean_text(text: str) -> str:
"""
Basic text cleaning:
- Lowercase
- Remove URLs, mentions, hashtags
- Remove numbers, punctuation, and stopwords
"""
text = text.lower()
# Remove URLs
text = re.sub(r"http\S+|www\S+|https\S+", "", text)
# Remove mentions and hashtags
text = re.sub(r"@\w+|#\w+", "", text)
# Remove numbers and special characters
text = re.sub(r"[^a-z\s]", "", text)
# Remove stopwords
tokens = [word for word in text.split() if word not in STOPWORDS]
return " ".join(tokens)
def preprocess_texts(texts: list) -> list:
"""
Clean and preprocess a list of texts.
"""
return [clean_text(t) for t in texts if t.strip()]