Arabic-Sentiment-Analysis / data /preparation.py
mhamza-007's picture
Upload 8 files
0104e91 verified
Raw
History Blame Contribute Delete
2.5 kB
"""Shaping the raw DataFrame: label mapping, deduplication and balancing."""
import pandas as pd
from config.constants import (
LABEL_COLUMN,
RATING_COLUMN,
RATING_TO_SENTIMENT,
RAW_COLUMNS_TO_DROP,
RAW_TEXT_COLUMN_CANDIDATES,
SENTIMENT_COLUMN,
TEXT_COLUMN,
)
def resolve_raw_text_column(df, text_column=None):
"""Return the name of the raw text column present in ``df``."""
if text_column is not None:
return text_column
for candidate in RAW_TEXT_COLUMN_CANDIDATES:
if candidate in df.columns:
return candidate
raise KeyError(
'No raw text column found; expected one of %s, got %s'
% (list(RAW_TEXT_COLUMN_CANDIDATES), list(df.columns))
)
def prepare_dataframe(df, text_column=None):
"""Drop bookkeeping columns, map ratings to sentiments and rename columns."""
text_column = resolve_raw_text_column(df, text_column)
df = df.drop(RAW_COLUMNS_TO_DROP, axis=1)
# Replace numeric values in the 'Rating' column with corresponding sentiment labels
df[SENTIMENT_COLUMN] = df[RATING_COLUMN].replace(RATING_TO_SENTIMENT)
# Rename columns
df = df.rename(columns={text_column: TEXT_COLUMN})
df = df.drop(RATING_COLUMN, axis=1)
return df
def rename_sentiment_to_label(df):
"""Rename the ``Sentiment`` column to ``label``."""
return df.rename(columns={SENTIMENT_COLUMN: LABEL_COLUMN})
def drop_missing_and_duplicates(df):
"""Drop rows with missing values and rows whose text is a duplicate."""
df = df.dropna()
index = df[df[TEXT_COLUMN].duplicated() == True].index
df.drop(index, axis=0, inplace=True)
return df
def balance_labels(df, label_column=LABEL_COLUMN, random_state=None):
"""Downsample the majority labels to the size of the smallest one.
Returns:
A new, index-reset, class-balanced DataFrame.
"""
sentiment_counts = df[label_column].value_counts()
target_count = sentiment_counts.min()
# Downsample 'Neutral' and 'Negative' to match the target count
neutral_equalized = df[df[label_column] == 'Neutral'].sample(
target_count, replace=False, random_state=random_state)
negative_equalized = df[df[label_column] == 'Negative'].sample(
target_count, replace=False, random_state=random_state)
# Combine the equalized dataframes
return pd.concat(
[neutral_equalized, negative_equalized, df[df[label_column] == 'Positive']]
).reset_index(drop=True)