mhamza-007's picture
Upload 56 files
d840583 verified
Raw
History Blame Contribute Delete
1.46 kB
"""Sentiment prediction for new, unseen text."""
import pandas as pd
from config.constants import TEXT_COLUMN
from config.paths import DEFAULT_MODEL_PATH
from preprocessing.pipeline import preprocess_dataframe
from training.model_io import load_model
_default_model = None
def get_default_model(path=DEFAULT_MODEL_PATH):
"""Load (and cache) the saved best model."""
global _default_model
if _default_model is None:
_default_model = load_model(path)
return _default_model
def predict_sentiment(text, model=None, convert_emojis=True):
"""Predict the sentiment of a single piece of raw text.
The text goes through the same cleaning chain the training corpus went
through - emoticons to emojis, emojis to Arabic words, remove stopwords,
remove non-Arabic characters, normalize, remove numbers, remove
hashtags/mentions, remove URLs, remove punctuation, light-stem - and is
then handed to the fitted TF-IDF + classifier pipeline.
"""
if model is None:
model = get_default_model()
# Create a DataFrame with the input text
text = [text]
daf = pd.DataFrame({TEXT_COLUMN: text})
# Preprocess the text
daf = preprocess_dataframe(daf, text_column=TEXT_COLUMN,
convert_emojis=convert_emojis,
drop_small_sentences=False)
# Make predictions
predictions = model.predict(daf[TEXT_COLUMN])
return predictions