| |
| """Untitled2.ipynb |
| |
| Automatically generated by Colab. |
| |
| Original file is located at |
| https://colab.research.google.com/drive/1HGae1uLkMV49QG1y0yu-3y6RaTi_koeo |
| """ |
|
|
| |
| import pandas as pd |
| import numpy as np |
| from sklearn.feature_extraction.text import TfidfVectorizer |
| from sklearn.metrics.pairwise import cosine_similarity |
| import re |
| import random |
| import gradio as gr |
| import nltk |
| from nltk.corpus import stopwords |
| from nltk.stem import WordNetLemmatizer |
|
|
| |
| nltk.download('punkt') |
| nltk.download('wordnet') |
| nltk.download('stopwords') |
|
|
| class CustomerSupportChatbot: |
| def __init__(self, data_path='sample.csv'): |
| """Initialize the chatbot with data""" |
| self.df = self.load_data(data_path) |
| self.vectorizer = TfidfVectorizer(tokenizer=nltk.word_tokenize, preprocessor=self.preprocess_text) |
| self.tfidf_matrix = self.vectorizer.fit_transform(self.df['clean_text']) |
|
|
| |
| self.lemmatizer = WordNetLemmatizer() |
| self.stop_words = set(stopwords.words('english')) |
|
|
| |
| self.company_handles = { |
| 'AppleSupport', 'ChaseSupport', 'VirginTrains', 'SpotifyCares', |
| 'British_Airways', 'O2', 'comcastcares', 'sprintcare', |
| 'Ask_Spectrum', 'Tesco', 'SouthwestAir', 'HPSupport', 'UPSHelp' |
| } |
|
|
| def load_data(self, filepath): |
| """Load and preprocess the CSV data""" |
| df = pd.read_csv(filepath) |
|
|
| |
| df['clean_text'] = df['text'].apply(lambda x: re.sub(r'@\w+', '', x)) |
| df['clean_text'] = df['clean_text'].apply(lambda x: re.sub(r'http\S+|www\S+|https\S+', '', x, flags=re.MULTILINE)) |
| df['clean_text'] = df['clean_text'].apply(lambda x: re.sub(r'\W', ' ', x)) |
| df['clean_text'] = df['clean_text'].apply(lambda x: x.lower()) |
|
|
| return df |
|
|
| def preprocess_text(self, text): |
| """Tokenize, remove stopwords, and lemmatize text""" |
| words = nltk.word_tokenize(text.lower()) |
| words = [self.lemmatizer.lemmatize(word) for word in words if word not in self.stop_words and word.isalpha()] |
| return ' '.join(words) |
|
|
| def get_response(self, query): |
| """Generate response to user query""" |
| try: |
| |
| clean_query = re.sub(r'@\w+', '', query) |
| clean_query = re.sub(r'http\S+|www\S+|https\S+', '', clean_query) |
| clean_query = re.sub(r'\W', ' ', clean_query) |
| clean_query = clean_query.lower() |
|
|
| |
| query_vec = self.vectorizer.transform([clean_query]) |
|
|
| |
| similarities = cosine_similarity(query_vec, self.tfidf_matrix).flatten() |
|
|
| |
| best_match_idx = similarities.argmax() |
| best_similarity = similarities[best_match_idx] |
|
|
| if best_similarity > 0.3: |
| response = self.df.iloc[best_match_idx]['text'] |
| author = self.df.iloc[best_match_idx]['author_id'] |
|
|
| |
| if not self.df.iloc[best_match_idx]['inbound']: |
| return f"{author}: {response}" |
| else: |
| |
| tweet_id = self.df.iloc[best_match_idx]['tweet_id'] |
| response_row = self.df[self.df['in_response_to_tweet_id'] == tweet_id] |
| if not response_row.empty: |
| return f"{response_row.iloc[0]['author_id']}: {response_row.iloc[0]['text']}" |
| else: |
| return self.get_fallback_response() |
| else: |
| return self.get_fallback_response() |
| except Exception as e: |
| print(f"Error generating response: {e}") |
| return "I'm having trouble understanding. Could you please rephrase your question?" |
|
|
| def get_fallback_response(self): |
| """Return a generic response when no good match is found""" |
| fallbacks = [ |
| "I'm sorry, I didn't quite understand that. Could you rephrase your question?", |
| "I'd be happy to help with that. Can you provide more details about your issue?", |
| "Thank you for your message. Let me connect you with the appropriate support team.", |
| "I'll do my best to assist you. What exactly seems to be the problem?", |
| "That's an important concern. Let me check our knowledge base for a solution.", |
| "I want to make sure I understand correctly. Could you explain your issue in more detail?", |
| "Thanks for reaching out! Let me find the best solution for your concern." |
| ] |
| return random.choice(fallbacks) |
|
|
| def launch_chat_interface(self): |
| """Launch the Gradio chat interface""" |
| def chat_with_bot(message, history): |
| return self.get_response(message) |
|
|
| iface = gr.ChatInterface( |
| fn=chat_with_bot, |
| title="Customer Support Chatbot", |
| description="Ask me about your customer service concerns. I'm trained on real support conversations.", |
| examples=[ |
| ["My iPhone battery drains too fast after update"], |
| ["Spotify keeps skipping songs"], |
| ["My flight was delayed"], |
| ["I'm having issues with my internet connection"], |
| ["The Tesco website isn't working"], |
| ["How do I contact Apple support?"] |
| ], |
| theme="soft", |
| css=".gradio-container {max-width: 800px; margin: auto;}" |
| ) |
|
|
| return iface |
|
|
| if __name__ == "__main__": |
| print("Initializing Customer Support Chatbot...") |
| chatbot = CustomerSupportChatbot() |
|
|
| print("Launching chat interface...") |
| iface = chatbot.launch_chat_interface() |
| iface.launch(share=True) |