import os import sys import numpy as np import pickle from pathlib import Path sys.path.append(str(Path(__file__).resolve().parent.parent.parent)) from src.config import TEXT_DATASET_PATH, MODELS_DIR, MENTAL_HEALTH_CATEGORIES from src.data_prep.generate_text_data import calculate_first_person_pronoun_ratio, calculate_negative_word_density try: import pandas as pd import torch from transformers import ( AutoTokenizer, AutoModelForSequenceClassification, Trainer, TrainingArguments, pipeline ) from datasets import Dataset from sklearn.metrics import accuracy_score, classification_report HAS_TRANSFORMERS = True except ImportError: HAS_TRANSFORMERS = False pd = None TEXT_MODEL_DIR = os.path.join(MODELS_DIR, "text_transformer") class LinguisticStressClassifier: """ Deep Learning Linguistic Stress Classifier using BERT. Fine-tunes a pre-trained transformer model on the mental health text dataset. """ def __init__(self, model_name="bert-base-uncased"): self.model_name = model_name self.classes_ = MENTAL_HEALTH_CATEGORIES self.num_labels = len(self.classes_) self.is_fitted = False if HAS_TRANSFORMERS: self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") self.tokenizer = AutoTokenizer.from_pretrained(self.model_name) self.model = None else: self.device = "cpu" self.tokenizer = None self.model = None def train_and_evaluate(self, data_path=TEXT_DATASET_PATH, test_data_path=None): if not HAS_TRANSFORMERS: print("[Text Pipeline] Transformers library not available. Skipping Deep Learning training.") return 0.0 if not os.path.exists(data_path): raise FileNotFoundError(f"Dataset not found at {data_path}. Run dataset generator first.") print(f"[Text Pipeline] Loading training dataset from {data_path}...") df_train = pd.read_csv(data_path) df_test = None if test_data_path and os.path.exists(test_data_path): print(f"[Text Pipeline] Loading testing dataset from {test_data_path}...") df_test = pd.read_csv(test_data_path) def process_labels(df): if "text" in df.columns: initial_len = len(df) df = df.drop_duplicates(subset=["text"]) print(f"[Text Pipeline] Dropped {initial_len - len(df)} duplicate rows.") y_int = [] for label in df["category"].values: label = str(label).strip() if label in self.classes_: y_int.append(self.classes_.index(label)) elif label == "Calm / Normal": y_int.append(self.classes_.index("Normal")) elif label in ["Academic Stress", "Non-Academic Stress", "Mixed Stress"]: y_int.append(self.classes_.index("Stress")) else: y_int.append(self.classes_.index("Normal")) df["label"] = y_int return df df_train = process_labels(df_train) from sklearn.model_selection import train_test_split if df_test is not None: df_test = process_labels(df_test) train_df = df_train test_df = df_test else: train_df, test_df = train_test_split(df_train, test_size=0.1, random_state=42, stratify=df_train["label"]) print(f"[Text Pipeline] Training samples: {len(train_df)}, Testing samples: {len(test_df)}") train_dataset = Dataset.from_pandas(train_df) test_dataset = Dataset.from_pandas(test_df) def tokenize_function(examples): return self.tokenizer(examples["text"], padding="max_length", truncation=True, max_length=128) print("[Text Pipeline] Tokenizing dataset for BERT...") tokenized_train = train_dataset.map(tokenize_function, batched=True) tokenized_test = test_dataset.map(tokenize_function, batched=True) self.model = AutoModelForSequenceClassification.from_pretrained( self.model_name, num_labels=self.num_labels, id2label={i: c for i, c in enumerate(self.classes_)}, label2id={c: i for i, c in enumerate(self.classes_)} ).to(self.device) training_args = TrainingArguments( output_dir=TEXT_MODEL_DIR, eval_strategy="epoch", save_strategy="epoch", learning_rate=5e-5, per_device_train_batch_size=16, per_device_eval_batch_size=16, num_train_epochs=8, weight_decay=0.01, save_total_limit=2, logging_dir='./logs', logging_steps=10, report_to="none" ) def compute_metrics(eval_pred): logits, labels = eval_pred predictions = np.argmax(logits, axis=-1) return {"accuracy": accuracy_score(labels, predictions)} trainer = Trainer( model=self.model, args=training_args, train_dataset=tokenized_train, eval_dataset=tokenized_test, compute_metrics=compute_metrics, ) print(f"[Text Pipeline] Starting Deep Learning Training on {self.device.type.upper()}...") from transformers.trainer_utils import get_last_checkpoint last_checkpoint = get_last_checkpoint(TEXT_MODEL_DIR) if os.path.isdir(TEXT_MODEL_DIR) else None if last_checkpoint is not None: print(f"[Text Pipeline] Resuming from checkpoint: {last_checkpoint}") trainer.train(resume_from_checkpoint=last_checkpoint) else: trainer.train() self.is_fitted = True print("[Text Pipeline] Evaluating Deep Learning Model...") eval_results = trainer.evaluate() acc = eval_results.get("eval_accuracy", 0.0) print(f"\n[Text Pipeline] BERT Test Accuracy: {acc * 100:.2f}%\n") self.save_model() return acc def predict(self, text): if not self.is_fitted: try: self.load_model() except Exception: pass if not self.is_fitted or self.model is None or not HAS_TRANSFORMERS: return self._heuristic_predict(text) pipe = pipeline("text-classification", model=self.model, tokenizer=self.tokenizer, device=0 if self.device.type=="cuda" else -1, top_k=None) results = pipe(str(text))[0] prob_dict = {res['label']: round(float(res['score']), 4) for res in results} pred_category = max(prob_dict, key=prob_dict.get) confidence = prob_dict[pred_category] calm_prob = prob_dict.get("Normal", 0.0) stress_prob = 1.0 - calm_prob neg_density = calculate_negative_word_density(text) stress_score = round(min(100.0, max(0.0, (stress_prob * 80.0) + (neg_density * 100.0))), 2) return { "predicted_category": pred_category, "probabilities": prob_dict, "linguistic_stress_score": stress_score, "confidence": confidence, "metadata": { "first_person_ratio": calculate_first_person_pronoun_ratio(text), "negative_word_density": calculate_negative_word_density(text), "word_count": len(str(text).split()) } } def save_model(self): if not HAS_TRANSFORMERS or self.model is None: return os.makedirs(TEXT_MODEL_DIR, exist_ok=True) self.model.save_pretrained(TEXT_MODEL_DIR) self.tokenizer.save_pretrained(TEXT_MODEL_DIR) print(f"[Text Pipeline] Transformer Model saved successfully to {TEXT_MODEL_DIR}") def load_model(self): if not HAS_TRANSFORMERS: raise ImportError("transformers not available") path = TEXT_MODEL_DIR if not os.path.exists(path) or not os.path.exists(os.path.join(path, "config.json")): alt_path = Path("/var/task/models_bin/text_transformer") if alt_path.exists(): path = str(alt_path) else: alt_path2 = Path(__file__).resolve().parent.parent.parent / "models_bin" / "text_transformer" if alt_path2.exists(): path = str(alt_path2) if os.path.exists(path) and os.path.exists(os.path.join(path, "config.json")): self.model = AutoModelForSequenceClassification.from_pretrained(path).to(self.device) self.tokenizer = AutoTokenizer.from_pretrained(path) self.is_fitted = True print(f"[Text Pipeline] Transformer loaded from {path}") else: raise FileNotFoundError(f"Deep learning model not found at {path}") def _heuristic_predict(self, text): t_low = str(text).lower() words = t_low.split() word_count = len(words) fp_ratio = calculate_first_person_pronoun_ratio(text) neg_density = calculate_negative_word_density(text) # Enhanced keywords for all 5 categories acad_hits = sum(1 for w in words if any(k in w for k in ["exam", "test", "deadline", "grade", "fail", "pass", "study", "studying", "class", "assignment"])) # Depression keywords depress_hits = sum(1 for w in words if any(k in w for k in ["depress", "hopeless", "emptiness", "worthless", "meaningless", "sad", "sadness", "give up"])) # Anxiety keywords anxiety_hits = sum(1 for w in words if any(k in w for k in ["anxi", "panic", "terrified", "worry", "worried", "scared", "fear", "nervous"])) # Emotional Distress / Trauma keywords distress_hits = sum(1 for w in words if any(k in w for k in ["grief", "trauma", "distress", "crying", "breakdown", "unbearable", "overwhelm", "pain", "hurt"])) raw_score = (neg_density * 120.0) + (fp_ratio * 35.0) + (acad_hits * 14.0) + (depress_hits * 20.0) + (anxiety_hits * 20.0) + (distress_hits * 25.0) stress_score = round(min(96.0, max(5.0, raw_score)), 2) # Determine category based on strongest hits if distress_hits > 0 and distress_hits >= max(depress_hits, anxiety_hits): pred_cat = "Emotional Distress" prob_dict = {"Normal": 0.05, "Stress": 0.15, "Depression": 0.1, "Anxiety": 0.1, "Emotional Distress": 0.6} stress_score = max(stress_score, 85.0) elif depress_hits > 0 and depress_hits >= anxiety_hits: pred_cat = "Depression" prob_dict = {"Normal": 0.05, "Stress": 0.15, "Depression": 0.6, "Anxiety": 0.1, "Emotional Distress": 0.1} stress_score = max(stress_score, 75.0) elif anxiety_hits > 0: pred_cat = "Anxiety" prob_dict = {"Normal": 0.05, "Stress": 0.15, "Depression": 0.1, "Anxiety": 0.6, "Emotional Distress": 0.1} stress_score = max(stress_score, 75.0) elif stress_score < 25.0 and acad_hits == 0 and neg_density < 0.05: pred_cat = "Normal" prob_dict = {"Normal": 0.82, "Stress": 0.18, "Depression": 0.0, "Anxiety": 0.0, "Emotional Distress": 0.0} else: pred_cat = "Stress" prob_dict = {"Normal": 0.18, "Stress": 0.82, "Depression": 0.0, "Anxiety": 0.0, "Emotional Distress": 0.0} return { "predicted_category": pred_cat, "probabilities": prob_dict, "linguistic_stress_score": stress_score, "confidence": round(float(prob_dict[pred_cat]), 4), "metadata": { "first_person_ratio": fp_ratio, "negative_word_density": neg_density, "word_count": word_count } } if __name__ == "__main__": classifier = LinguisticStressClassifier() classifier.train_and_evaluate()