import gradio as gr import pandas as pd import numpy as np import difflib import plotly.graph_objects as go from sklearn.feature_extraction.text import TfidfVectorizer from sklearn.metrics.pairwise import cosine_similarity from sklearn.preprocessing import MinMaxScaler # Load dataset with proper error handling def load_data(file_path='games_march2025_cleaned.csv', max_rows=88899): try: data = pd.read_csv(file_path, quotechar='"', on_bad_lines='skip', nrows=max_rows) print(f"Successfully loaded {len(data)} games from {file_path}") return data except Exception as e: print(f"Error loading data: {e}") # Return empty DataFrame with expected columns to avoid crashing return pd.DataFrame(columns=['name', 'genres', 'categories', 'tags', 'platforms', 'positive_ratings', 'price']) # Load and preprocess data data = load_data() # Only proceed if we have data if len(data) > 0: # Handle missing values for feature in ['genres', 'categories', 'tags', 'platforms', 'positive_ratings', 'negative_ratings', 'price']: if feature not in data.columns: data[feature] = '' elif data[feature].dtype == object: # String columns data[feature] = data[feature].fillna('') else: data[feature] = data[feature].fillna(0) # Numeric columns # Add derived features for better recommendations if 'positive_ratings' in data.columns and 'negative_ratings' in data.columns: data['rating_ratio'] = data['positive_ratings'] / (data['positive_ratings'] + data['negative_ratings'] + 1) else: data['rating_ratio'] = 0.5 # Default neutral rating # Add playtime features if available if 'average_playtime_forever' in data.columns: # Log transform to handle skewed distribution data['log_playtime'] = np.log1p(data['average_playtime_forever']) scaler = MinMaxScaler() data['playtime_scaled'] = scaler.fit_transform(data[['log_playtime']]) else: data['playtime_scaled'] = 0.5 # Add user score features if available if 'user_score' in data.columns: data['user_score_scaled'] = data['user_score'] / 100.0 # Assuming user_score is out of 100 else: data['user_score_scaled'] = 0.5 # Create a more comprehensive combined feature set with weighted components data['combined_features'] = '' # Add name with higher weight for better keyword matching if 'name' in data.columns: data['combined_features'] += data['name'].astype(str) + ' ' + data['name'].astype(str) + ' ' # Add genres with higher weight (repeat to increase importance) if 'genres' in data.columns: data['combined_features'] += data['genres'].astype(str) + ' ' + data['genres'].astype(str) + ' ' # Add other features for feature in ['categories', 'tags', 'platforms']: if feature in data.columns: data['combined_features'] += data[feature].astype(str) + ' ' # Add developers and publishers if available for feature in ['developers', 'publishers']: if feature in data.columns: data['combined_features'] += data[feature].astype(str) + ' ' # Clean the combined features data['combined_features'] = data['combined_features'].str.replace(';', ' ').str.replace("'", '').str.replace('[', '').str.replace(']', '').str.replace('{', '').str.replace('}', '').str.lower() # Vectorize with improved parameters try: # Use more n-grams and increased max_features for better semantic understanding vectorizer = TfidfVectorizer( stop_words='english', ngram_range=(1, 3), # Capture phrases up to 3 words max_features=10000, # Increase features for more nuanced relationships min_df=2, # Ignore very rare terms max_df=0.9 # Ignore very common terms ) feature_vectors = vectorizer.fit_transform(data['combined_features']) print(f"Vectorization complete. Shape: {feature_vectors.shape}") except Exception as e: print(f"Vectorization error: {e}") feature_vectors = np.zeros((len(data), 1)) # Normalize ratings with sigmoid-like scaling for better differentiation if 'positive_ratings' in data.columns and len(data) > 0: # Log transform to handle skewed distribution of ratings data['log_ratings'] = np.log1p(data['positive_ratings']) scaler = MinMaxScaler() data['positive_ratings_scaled'] = scaler.fit_transform(data[['log_ratings']]) else: data['positive_ratings_scaled'] = 0 # Compute similarity matrix with optimizations if feature_vectors.shape[0] > 1: try: if len(data) > 5000: print("Large dataset detected. Using batched similarity calculation.") batch_size = 1000 similarity_matrix = np.zeros((len(data), len(data))) for i in range(0, len(data), batch_size): end = min(i + batch_size, len(data)) batch = feature_vectors[i:end] similarity_matrix[i:end] = cosine_similarity(batch, feature_vectors) game_similarity = similarity_matrix else: game_similarity = cosine_similarity(feature_vectors) print(f"Similarity matrix created. Shape: {game_similarity.shape}") except Exception as e: print(f"Similarity calculation error: {e}") game_similarity = np.eye(len(data)) else: game_similarity = np.eye(len(data)) list_of_all_titles = data['name'].tolist() else: feature_vectors = np.zeros((0, 0)) game_similarity = np.zeros((0, 0)) list_of_all_titles = [] # Improved platform detection function def detect_platforms(platforms_str): platforms = [] if isinstance(platforms_str, str): platforms_str = platforms_str.lower() if 'windows' in platforms_str or 'true' in platforms_str: platforms.append("Windows") if any(mac_term in platforms_str for mac_term in ['mac', 'macos', 'osx']): platforms.append("macOS") if 'linux' in platforms_str: platforms.append("Linux") if any(mobile_term in platforms_str for mobile_term in ['android', 'ios', 'mobile']): platforms.append("Mobile") elif isinstance(platforms_str, bool) and platforms_str: # Handle boolean True values platforms.append("Windows") # Assuming Windows by default if boolean True return platforms if platforms else ["Unknown"] # Extract platform information from dataset columns def get_platforms(row): platforms = [] # Check for platform columns from the screenshots (windows, mac, linux) if 'windows' in row and row['windows']: platforms.append("Windows") if 'mac' in row and row['mac']: platforms.append("macOS") if 'linux' in row and row['linux']: platforms.append("Linux") # If no platforms detected but there's a platforms field, try that if not platforms and 'platforms' in row: platforms = detect_platforms(row['platforms']) return platforms if platforms else ["Unknown"] # Extract genre information def extract_genres(genres_str): if not genres_str or pd.isna(genres_str): return [] # Handle different formats that might be in the data if isinstance(genres_str, str): # Remove common formatting characters clean_str = genres_str.replace("'", "").replace("[", "").replace("]", "").replace("{", "").replace("}", "") # Try different delimiters if ',' in clean_str: return [g.strip() for g in clean_str.split(',') if g.strip()] elif ';' in clean_str: return [g.strip() for g in clean_str.split(';') if g.strip()] else: return [clean_str] return [] # Create price gauge visualization def create_price_gauge(game_price, similar_games_prices): # Add the main game price to the list all_prices = [game_price] + similar_games_prices # Filter out None values and convert to float all_prices = [float(p) if p is not None else 0 for p in all_prices] # Calculate stats max_price = max(all_prices) if all_prices else 60 # Default max if no prices avg_price = sum(all_prices) / len(all_prices) if all_prices else 0 # Create gauge for the main game price fig = go.Figure(go.Indicator( mode="gauge+number", value=game_price if game_price is not None else 0, title={'text': "Game Price ($)"}, gauge={ 'axis': {'range': [0, max(max_price, 60)]}, # Ensure reasonable scale 'bar': {'color': "#1DB954"}, # Steam-like green 'steps': [ {'range': [0, avg_price], 'color': "lightgray"}, {'range': [avg_price, max_price], 'color': "gray"} ], 'threshold': { 'line': {'color': "red", 'width': 4}, 'thickness': 0.75, 'value': avg_price } } )) fig.update_layout( height=300, margin=dict(l=20, r=20, t=50, b=20), ) return fig # Create user ratings visualization def create_ratings_chart(game_data): if not isinstance(game_data, dict): return None # Extract ratings data game_name = game_data.get('name', 'Unknown') positive = game_data.get('positive', 0) negative = game_data.get('negative', 0) # Calculate percentages total = positive + negative if total == 0: positive_pct = 0 negative_pct = 0 else: positive_pct = (positive / total) * 100 negative_pct = (negative / total) * 100 # Create bar chart fig = go.Figure() fig.add_trace(go.Bar( x=['Positive', 'Negative'], y=[positive, negative], text=[f"{positive:,} ({positive_pct:.1f}%)", f"{negative:,} ({negative_pct:.1f}%)"], textposition='auto', marker_color=['#66c0f4', '#ff7b7b'] # Steam-like colors )) fig.update_layout( title=f"User Ratings for {game_name}", xaxis_title="Rating Type", yaxis_title="Number of Ratings", height=300, margin=dict(l=20, r=20, t=50, b=20), ) return fig # Enhanced game recommendation function def recommend_games(user_game_name_input): if not user_game_name_input or not list_of_all_titles: return "Please enter a game name and ensure the dataset is loaded.", [], None, None # Normalize input for better matching user_input_cleaned = user_game_name_input.strip().lower() # First try exact match (case insensitive) exact_matches = [title for title in list_of_all_titles if title.lower() == user_input_cleaned] if exact_matches: closest_match = exact_matches[0] else: # Try partial match before fuzzy matching partial_matches = [title for title in list_of_all_titles if user_input_cleaned in title.lower()] if partial_matches: # Sort by length to prefer shorter (more exact) matches closest_match = sorted(partial_matches, key=len)[0] else: # Try fuzzy matching with improved parameters find_close_match = difflib.get_close_matches( user_game_name_input, list_of_all_titles, n=5, # Get more candidates cutoff=0.5 # Lower threshold for more possibilities ) if not find_close_match: return f"No match found for '{user_game_name_input}'. Please try another game name.", [], None, None # Take the closest match closest_match = find_close_match[0] try: index_of_the_game = data.loc[data['name'] == closest_match].index[0] # Check for valid index if index_of_the_game >= len(game_similarity): return f"Found match '{closest_match}' but encountered an indexing error.", [], None, None similarity_scores = list(enumerate(game_similarity[index_of_the_game])) # Enhanced ranking with hybrid scoring game_rankings = [] for idx, sim_score in similarity_scores: if idx == index_of_the_game: # Skip the game itself continue # Get additional factors for hybrid scoring rating_factor = data.iloc[idx]['positive_ratings_scaled'] # Calculate genre similarity separately searched_game_genres = extract_genres(data.iloc[index_of_the_game].get('genres', '')) current_game_genres = extract_genres(data.iloc[idx].get('genres', '')) # Count matching genres matching_genres = len(set(searched_game_genres) & set(current_game_genres)) genre_factor = matching_genres / max(len(searched_game_genres), 1) if searched_game_genres else 0 # Add playtime score if available playtime_factor = data.iloc[idx].get('playtime_scaled', 0) # Create hybrid score with weights hybrid_score = ( 0.60 * sim_score + # Base similarity from TF-IDF vectors 0.15 * rating_factor + # Rating popularity 0.15 * genre_factor + # Genre match 0.10 * playtime_factor # Playtime popularity ) game_rankings.append((idx, hybrid_score)) # Sort by the hybrid score sorted_similar_games = sorted(game_rankings, key=lambda x: x[1], reverse=True) recommendations = [] game_list = [] # Get searched game details searched_game = data.iloc[index_of_the_game] # Extract genres searched_game_genres = extract_genres(searched_game.get('genres', '')) searched_game_genres_display = ", ".join([g for g in searched_game_genres if g]) # Extract platforms searched_game_platforms = get_platforms(searched_game) searched_game_platform_display = ", ".join(searched_game_platforms) # Get price searched_game_price = searched_game.get('price', 0) searched_game_price_display = f"${searched_game_price:.2f}" if isinstance(searched_game_price, (int, float)) else "N/A" # Get ratings searched_game_positive = searched_game.get('positive_ratings', searched_game.get('positive', 0)) searched_game_negative = searched_game.get('negative_ratings', searched_game.get('negative', 0)) # Get metacritic score if available metacritic_score = searched_game.get('metacritic_score', 'N/A') metacritic_display = f"{metacritic_score}/100" if metacritic_score != 'N/A' else "N/A" # Get user score if available user_score = searched_game.get('user_score', 'N/A') user_score_display = f"{user_score}/100" if user_score != 'N/A' else "N/A" # Get playtime if available avg_playtime = searched_game.get('average_playtime_forever', 0) playtime_display = f"{avg_playtime} minutes" if avg_playtime > 0 else "N/A" # Format the searched game with clean styling recommendations.append(f"## You searched for: {closest_match}\n" + f"**Genres:** {searched_game_genres_display}\n" + f"**Platforms:** {searched_game_platform_display}\n" + f"**Price:** {searched_game_price_display}\n" + f"**Metacritic Score:** {metacritic_display}\n" + f"**User Score:** {user_score_display}\n" + f"**Average Playtime:** {playtime_display}\n") game_list.append(closest_match) # Add a divider recommendations.append("---\n## Top Recommendations\n") # Get prices and ratings for similar games (for gauge visualization) similar_games_prices = [] # Create ratings data for visualization ratings_data = { 'name': closest_match, 'positive': searched_game_positive, 'negative': searched_game_negative } # Process recommendations with diversity enforcement seen_publishers = set() if 'publishers' in data.columns: searched_game_publisher = str(searched_game.get('publishers', '')).lower() seen_publishers.add(searched_game_publisher) recommended_count = 0 # Process recommendations for i, (index, score) in enumerate(sorted_similar_games): if score < 0.10: # Minimum threshold for quality continue # Enforce diversity by limiting games from same publisher if 'publishers' in data.columns: current_publisher = str(data.iloc[index].get('publishers', '')).lower() if current_publisher in seen_publishers and len(seen_publishers) > 2: continue seen_publishers.add(current_publisher) game_name = data.iloc[index]['name'] # Get platform info platform_list = get_platforms(data.iloc[index]) platform_display = ", ".join(platform_list) # Get price info price = data.iloc[index].get('price', 0) similar_games_prices.append(price) price_display = f"${price:.2f}" if isinstance(price, (int, float)) else "N/A" # Get genre info genres = extract_genres(data.iloc[index].get('genres', '')) genres_display = ", ".join([g for g in genres if g]) # Get metacritic score if available rec_metacritic_score = data.iloc[index].get('metacritic_score', 'N/A') rec_metacritic_display = f"{rec_metacritic_score}/100" if rec_metacritic_score != 'N/A' else "N/A" # Get user score if available rec_user_score = data.iloc[index].get('user_score', 'N/A') rec_user_score_display = f"{rec_user_score}/100" if rec_user_score != 'N/A' else "N/A" # Calculate match percentage match_percentage = min(int(score * 100), 100) # Cap at 100% # Format recommendation with clean styling position = recommended_count + 1 recommendation = ( f"### {position}. {game_name}\n" + f"**Match:** {match_percentage}%\n" + f"**Genres:** {genres_display}\n" + f"**Platforms:** {platform_display}\n" + f"**Price:** {price_display}\n" + f"**Metacritic Score:** {rec_metacritic_display}\n" + f"**User Score:** {rec_user_score_display}\n" ) recommendations.append(recommendation) game_list.append(game_name) recommended_count += 1 if recommended_count >= 5: # Stop after 5 recommendations break # Create price gauge visualization price_gauge = create_price_gauge(searched_game_price, similar_games_prices) # Create ratings chart ratings_chart = create_ratings_chart(ratings_data) return "\n".join(recommendations), game_list, price_gauge, ratings_chart except Exception as e: return f"Error while finding recommendations: {str(e)}", [], None, None # Gradio UI with improved design with gr.Blocks(theme=gr.themes.Soft()) as demo: gr.Markdown("# Steam Game Recommender") gr.Markdown("Enter the name of a game you like and get recommendations based on similarity analysis of our Steam games dataset.") with gr.Row(): with gr.Column(scale=4): input_box = gr.Textbox( label="Your Favorite Game", placeholder="e.g., Counter-Strike, PUBG, Dota 2, Grand Theft Auto V", info="Type a game name that exists in the Steam dataset" ) with gr.Column(scale=1): run_button = gr.Button("Find Recommendations", variant="primary") with gr.Tabs(): with gr.TabItem("Recommendations"): with gr.Row(): with gr.Column(scale=3): # Recommendations output output_text = gr.Markdown(label="Recommendations") with gr.Column(scale=2): with gr.Row(): # Price gauge visualization price_gauge = gr.Plot(label="Price Comparison") with gr.Row(): # Ratings chart ratings_chart = gr.Plot(label="User Ratings") with gr.TabItem("About"): gr.Markdown(""" ## About This Recommender This Steam game recommender system uses machine learning to find games similar to your favorites. It analyzes: - Game genres and categories - User ratings and reviews - Platform availability - Tags and game descriptions - Price points - Player statistics The recommendations are based on a hybrid scoring system that combines content similarity, user ratings, and gameplay metrics. For best results, enter the exact name of a game that exists in the Steam database. """) # Register event def on_submit(user_input): rec_text, game_list, gauge, ratings = recommend_games(user_input) return rec_text, gauge, ratings run_button.click( fn=on_submit, inputs=input_box, outputs=[output_text, price_gauge, ratings_chart], show_progress=True ) # Also trigger on Enter key input_box.submit( fn=on_submit, inputs=input_box, outputs=[output_text, price_gauge, ratings_chart], show_progress=True ) # Launch the Gradio app if __name__ == "__main__": demo.launch()