# -*- coding: utf-8 -*- """Untitled2.ipynb Automatically generated by Colab. Original file is located at https://colab.research.google.com/drive/1HGae1uLkMV49QG1y0yu-3y6RaTi_koeo """ # customer_support_chatbot.py import pandas as pd import numpy as np from sklearn.feature_extraction.text import TfidfVectorizer from sklearn.metrics.pairwise import cosine_similarity import re import random import gradio as gr import nltk from nltk.corpus import stopwords from nltk.stem import WordNetLemmatizer # Download NLTK resources nltk.download('punkt') nltk.download('wordnet') nltk.download('stopwords') class CustomerSupportChatbot: def __init__(self, data_path='sample.csv'): """Initialize the chatbot with data""" self.df = self.load_data(data_path) self.vectorizer = TfidfVectorizer(tokenizer=nltk.word_tokenize, preprocessor=self.preprocess_text) self.tfidf_matrix = self.vectorizer.fit_transform(self.df['clean_text']) # Initialize NLP components self.lemmatizer = WordNetLemmatizer() self.stop_words = set(stopwords.words('english')) # Common company handles from the data self.company_handles = { 'AppleSupport', 'ChaseSupport', 'VirginTrains', 'SpotifyCares', 'British_Airways', 'O2', 'comcastcares', 'sprintcare', 'Ask_Spectrum', 'Tesco', 'SouthwestAir', 'HPSupport', 'UPSHelp' } def load_data(self, filepath): """Load and preprocess the CSV data""" df = pd.read_csv(filepath) # Clean text data df['clean_text'] = df['text'].apply(lambda x: re.sub(r'@\w+', '', x)) # Remove mentions df['clean_text'] = df['clean_text'].apply(lambda x: re.sub(r'http\S+|www\S+|https\S+', '', x, flags=re.MULTILINE)) # Remove URLs df['clean_text'] = df['clean_text'].apply(lambda x: re.sub(r'\W', ' ', x)) # Remove special chars df['clean_text'] = df['clean_text'].apply(lambda x: x.lower()) # Lowercase return df def preprocess_text(self, text): """Tokenize, remove stopwords, and lemmatize text""" words = nltk.word_tokenize(text.lower()) words = [self.lemmatizer.lemmatize(word) for word in words if word not in self.stop_words and word.isalpha()] return ' '.join(words) def get_response(self, query): """Generate response to user query""" try: # Preprocess query clean_query = re.sub(r'@\w+', '', query) clean_query = re.sub(r'http\S+|www\S+|https\S+', '', clean_query) clean_query = re.sub(r'\W', ' ', clean_query) clean_query = clean_query.lower() # Vectorize query query_vec = self.vectorizer.transform([clean_query]) # Calculate similarity similarities = cosine_similarity(query_vec, self.tfidf_matrix).flatten() # Get most similar response best_match_idx = similarities.argmax() best_similarity = similarities[best_match_idx] if best_similarity > 0.3: # Threshold for matching response = self.df.iloc[best_match_idx]['text'] author = self.df.iloc[best_match_idx]['author_id'] # Check if it's a company response or customer message if not self.df.iloc[best_match_idx]['inbound']: return f"{author}: {response}" else: # Find the company response to this customer message tweet_id = self.df.iloc[best_match_idx]['tweet_id'] response_row = self.df[self.df['in_response_to_tweet_id'] == tweet_id] if not response_row.empty: return f"{response_row.iloc[0]['author_id']}: {response_row.iloc[0]['text']}" else: return self.get_fallback_response() else: return self.get_fallback_response() except Exception as e: print(f"Error generating response: {e}") return "I'm having trouble understanding. Could you please rephrase your question?" def get_fallback_response(self): """Return a generic response when no good match is found""" fallbacks = [ "I'm sorry, I didn't quite understand that. Could you rephrase your question?", "I'd be happy to help with that. Can you provide more details about your issue?", "Thank you for your message. Let me connect you with the appropriate support team.", "I'll do my best to assist you. What exactly seems to be the problem?", "That's an important concern. Let me check our knowledge base for a solution.", "I want to make sure I understand correctly. Could you explain your issue in more detail?", "Thanks for reaching out! Let me find the best solution for your concern." ] return random.choice(fallbacks) def launch_chat_interface(self): """Launch the Gradio chat interface""" def chat_with_bot(message, history): return self.get_response(message) iface = gr.ChatInterface( fn=chat_with_bot, title="Customer Support Chatbot", description="Ask me about your customer service concerns. I'm trained on real support conversations.", examples=[ ["My iPhone battery drains too fast after update"], ["Spotify keeps skipping songs"], ["My flight was delayed"], ["I'm having issues with my internet connection"], ["The Tesco website isn't working"], ["How do I contact Apple support?"] ], theme="soft", css=".gradio-container {max-width: 800px; margin: auto;}" ) return iface if __name__ == "__main__": print("Initializing Customer Support Chatbot...") chatbot = CustomerSupportChatbot() print("Launching chat interface...") iface = chatbot.launch_chat_interface() iface.launch(share=True) # Set share=True to get a public URL