import os import pandas as pd from datasets import load_dataset import sys from pathlib import Path # Add project root to path sys.path.append(str(Path(__file__).resolve().parent.parent.parent)) from src.config import TEXT_DATASET_PATH, DATA_DIR def download_goemotions(): print("[Dataset Expansion] Downloading GoEmotions dataset (43,410 text samples)...") try: # Load the dataset from HuggingFace using the full namespace to prevent URI errors dataset = load_dataset("google-research-datasets/go_emotions", "simplified") train_data = dataset["train"].to_pandas() print("[Dataset Expansion] GoEmotions downloaded successfully. Processing...") # GoEmotions has 27 labels. We will map them to your system's Stress Categories # 0: admiration, 1: amusement, 2: anger, 3: annoyance, 4: approval, 5: caring, # 6: confusion, 7: curiosity, 8: desire, 9: disappointment, 10: disapproval, # 11: disgust, 12: embarrassment, 13: excitement, 14: fear, 15: gratitude, # 16: grief, 17: joy, 18: love, 19: nervousness, 20: optimism, 21: pride, # 22: realization, 23: relief, 24: remorse, 25: sadness, 26: surprise, 27: neutral def map_emotion_to_stress(label_list): if not len(label_list): return "Normal" primary_label = label_list[0] # Stress: anger, annoyance, disapproval, disgust if primary_label in [2, 3, 10, 11]: return "Stress" # Depression: disappointment, grief, remorse, sadness elif primary_label in [9, 16, 24, 25]: return "Depression" # Anxiety: fear, nervousness elif primary_label in [14, 19]: return "Anxiety" # Emotional Distress: confusion, curiosity, desire, embarrassment, excitement, realization, surprise elif primary_label in [6, 7, 8, 12, 13, 22, 26]: return "Emotional Distress" # Normal: admiration, amusement, approval, caring, gratitude, joy, love, optimism, pride, relief, neutral else: return "Normal" train_data['category'] = train_data['labels'].apply(map_emotion_to_stress) # Create final dataframe expanded_df = pd.DataFrame({ "text": train_data["text"], "category": train_data["category"] }) # Append to existing dataset or save as new if os.path.exists(TEXT_DATASET_PATH): existing_df = pd.read_csv(TEXT_DATASET_PATH) # Make sure we only append if columns match if "text" in existing_df.columns and "category" in existing_df.columns: final_df = pd.concat([existing_df, expanded_df], ignore_index=True) final_df = final_df.drop_duplicates(subset=["text"]) final_df.to_csv(TEXT_DATASET_PATH, index=False) print(f"[Dataset Expansion] SUCCESS! Expanded Text Dataset to {len(final_df)} rows. Saved to {TEXT_DATASET_PATH}") else: out_path = os.path.join(DATA_DIR, "massive_text_dataset.csv") expanded_df.to_csv(out_path, index=False) print(f"[Dataset Expansion] Saved new massive dataset to {out_path}") else: expanded_df.to_csv(TEXT_DATASET_PATH, index=False) print(f"[Dataset Expansion] Created new dataset at {TEXT_DATASET_PATH}") except Exception as e: print(f"[Dataset Expansion] Error downloading GoEmotions: {e}") def audio_dataset_instructions(): print("\n" + "="*80) print("AUDIO DATASET EXPANSION (CMU-MOSEI / DAIC-WOZ)") print("="*80) print("Due to strict academic Non-Disclosure Agreements (NDAs) and massive file sizes (60GB+),") print("you must manually request access to DAIC-WOZ and D-Vlog from their university creators.") print("\nTo use the massive open-source CMU-MOSEI dataset, use the CMU Multimodal SDK in Colab:") print("1. Run: !pip install mmsdk") print("2. In a Colab cell, use the following code to download it to your Drive:") print(" from mmsdk import mmdatasdk") print(" cmumosei_highlevel = mmdatasdk.mmdataset(mmdatasdk.cmu_mosei.highlevel, '/content/drive/MyDrive/NeuroSense_AI/data/MOSEI/')") print("="*80 + "\n") if __name__ == "__main__": download_goemotions() audio_dataset_instructions()