Spaces:
Running
Running
| """ | |
| 🎯 Clustering API - Unsupervised Machine Learning (Production AutoML) | |
| ===================================================================== | |
| Real AutoML Unsupervised Learning with: | |
| - Multiple algorithm comparison (KMeans, DBSCAN, GMM, Hierarchical, Spectral, K-Prototypes) | |
| - Automatic optimal k detection (elbow + silhouette) | |
| - Model persistence for predictions | |
| - Comprehensive visualization & charts | |
| - Production intelligence & reliability scoring | |
| 🛡️ PRODUCTION INTELLIGENCE INTEGRATED: | |
| - Data quality assessment | |
| - Feature validation | |
| - Missing data handling | |
| - Reliability scoring for cluster quality | |
| - Validation warnings | |
| """ | |
| from fastapi import APIRouter, HTTPException, UploadFile, File, Form, Header | |
| from pydantic import BaseModel | |
| from typing import Optional, Dict, Any, List | |
| import pandas as pd | |
| import numpy as np | |
| from sklearn.cluster import KMeans, DBSCAN, SpectralClustering, AgglomerativeClustering | |
| from sklearn.mixture import GaussianMixture | |
| from sklearn.preprocessing import StandardScaler | |
| from sklearn.decomposition import PCA | |
| from sklearn.metrics import silhouette_score, davies_bouldin_score, calinski_harabasz_score | |
| import io | |
| import logging | |
| import pickle | |
| import uuid | |
| import json | |
| from pathlib import Path | |
| from datetime import datetime | |
| import base64 | |
| # Import production clustering engine | |
| try: | |
| from ml.clustering_engine import ProductionClusteringEngine, clustering_engine | |
| except ImportError: | |
| clustering_engine = None | |
| # Import user paths utility | |
| try: | |
| from utils.paths import get_user_paths | |
| except ImportError: | |
| def get_user_paths(user_id): | |
| base = Path("storage/users") / user_id | |
| paths = {"base": base, "files": base / "files", "models": base / "models"} | |
| for p in paths.values(): | |
| p.mkdir(parents=True, exist_ok=True) | |
| return paths | |
| logger = logging.getLogger(__name__) | |
| router = APIRouter() | |
| # =========================================================================== | |
| # REQUEST/RESPONSE MODELS | |
| # =========================================================================== | |
| class ClusteringRequest(BaseModel): | |
| """Request model for clustering from file_id""" | |
| file_id: str | |
| user_id: Optional[str] = None | |
| algorithm: str = "auto" # auto, kmeans, dbscan, hierarchical, gmm, spectral | |
| n_clusters: Optional[int] = None # None = auto-detect | |
| compare_all: bool = False # Compare all algorithms | |
| exclude_columns: Optional[List[str]] = None | |
| class ClusterPredictRequest(BaseModel): | |
| """Request model for predicting cluster of new data point""" | |
| user_id: Optional[str] = None | |
| model_id: str | |
| features: Dict[str, float] | |
| # =========================================================================== | |
| # 🎯 MAIN CLUSTERING ENDPOINT - JSON API (Used by frontend) | |
| # =========================================================================== | |
| async def run_clustering_analysis( | |
| request: ClusteringRequest, | |
| x_user_id: Optional[str] = Header(None, alias="X-User-ID") | |
| ): | |
| """ | |
| 🎯 PRODUCTION AutoML Unsupervised Learning | |
| Takes a file_id and runs real clustering analysis with: | |
| - Automatic algorithm selection based on data type | |
| - Optimal k detection via silhouette analysis | |
| - Multiple algorithm comparison (optional) | |
| - Model persistence for predictions | |
| - Comprehensive visualizations | |
| Request Body: | |
| - file_id: User's file name/id | |
| - algorithm: 'auto', 'kmeans', 'dbscan', 'hierarchical', 'gmm', 'spectral' | |
| - n_clusters: Number of clusters (auto-detect if None) | |
| - compare_all: Run all algorithms and compare | |
| - exclude_columns: Columns to exclude from clustering | |
| """ | |
| user_id = request.user_id or x_user_id | |
| if not user_id: | |
| raise HTTPException(status_code=400, detail="User ID required") | |
| logger.info(f"🎯 Clustering request: user={user_id}, file={request.file_id}, algo={request.algorithm}") | |
| try: | |
| # 1. Load user's file | |
| paths = get_user_paths(user_id) | |
| files_dir = paths.get("files", paths["base"] / "files") | |
| # Try multiple file path patterns (CSV + Excel) | |
| file_path = None | |
| search_patterns = [ | |
| files_dir / request.file_id, | |
| files_dir / f"{request.file_id}.csv", | |
| files_dir / f"{request.file_id}.xlsx", | |
| files_dir / f"{request.file_id}.xls", | |
| Path(f"storage/users/{user_id}/files/{request.file_id}"), | |
| Path(f"storage/users/{user_id}/files/{request.file_id}.csv"), | |
| Path(f"storage/users/{user_id}/files/{request.file_id}.xlsx"), | |
| Path(f"storage/users/{user_id}/files/{request.file_id}.xls"), | |
| Path(f"backend/storage/users/{user_id}/files/{request.file_id}"), | |
| Path(f"backend/storage/users/{user_id}/files/{request.file_id}.csv"), | |
| Path(f"backend/storage/users/{user_id}/files/{request.file_id}.xlsx"), | |
| ] | |
| for pattern in search_patterns: | |
| if pattern.exists(): | |
| file_path = pattern | |
| break | |
| # Fallback: glob search in user's files directory for partial match | |
| if not file_path and files_dir.exists(): | |
| file_stem = Path(request.file_id).stem # Remove extension if present | |
| for ext in ['*.csv', '*.xlsx', '*.xls', '*.tsv']: | |
| matches = list(files_dir.glob(ext)) | |
| for m in matches: | |
| if file_stem.lower() in m.stem.lower() or m.name == request.file_id: | |
| file_path = m | |
| break | |
| if file_path: | |
| break | |
| # Last resort: just pick any data file if only one exists | |
| if not file_path: | |
| all_data_files = [f for f in files_dir.iterdir() | |
| if f.is_file() and f.suffix.lower() in ('.csv', '.xlsx', '.xls', '.tsv') | |
| and not f.name.startswith('cleaned_') and not f.name.startswith('clustered_')] | |
| if len(all_data_files) == 1: | |
| file_path = all_data_files[0] | |
| logger.info(f"📂 Auto-selected only available file: {file_path.name}") | |
| if not file_path: | |
| raise HTTPException( | |
| status_code=404, | |
| detail=f"File not found: {request.file_id}. Upload a file in DataHub first." | |
| ) | |
| logger.info(f"📂 Found file: {file_path}") | |
| # 2. Read data — support CSV and Excel | |
| file_ext = file_path.suffix.lower() | |
| if file_ext in ('.xlsx', '.xls'): | |
| df = pd.read_excel(file_path, engine='openpyxl' if file_ext == '.xlsx' else None) | |
| elif file_ext == '.tsv': | |
| df = pd.read_csv(file_path, sep='\t') | |
| else: | |
| df = pd.read_csv(file_path) | |
| logger.info(f"📊 Loaded {len(df)} rows x {len(df.columns)} columns from {file_ext}") | |
| if len(df) < 10: | |
| raise HTTPException(status_code=400, detail="Need at least 10 rows for clustering") | |
| # 3. Prepare exclude columns | |
| exclude_cols = request.exclude_columns or [] | |
| # Auto-exclude obvious ID/datetime columns | |
| for col in df.columns: | |
| col_lower = col.lower() | |
| if any(x in col_lower for x in ['id', 'index', 'key', 'timestamp', 'date', 'time', 'created', 'updated']): | |
| if col not in exclude_cols: | |
| exclude_cols.append(col) | |
| # 4. Run clustering with production engine or fallback | |
| result = await _run_production_clustering( | |
| df=df, | |
| algorithm=request.algorithm, | |
| n_clusters=request.n_clusters, | |
| exclude_columns=exclude_cols, | |
| compare_all=request.compare_all, | |
| user_id=user_id | |
| ) | |
| return result | |
| except HTTPException: | |
| raise | |
| except FileNotFoundError as e: | |
| raise HTTPException(status_code=404, detail=str(e)) | |
| except Exception as e: | |
| logger.error(f"❌ Clustering failed: {e}") | |
| import traceback | |
| traceback.print_exc() | |
| raise HTTPException(status_code=500, detail=f"Clustering error: {str(e)}") | |
| async def _run_production_clustering( | |
| df: pd.DataFrame, | |
| algorithm: str, | |
| n_clusters: Optional[int], | |
| exclude_columns: List[str], | |
| compare_all: bool, | |
| user_id: str | |
| ) -> Dict[str, Any]: | |
| """ | |
| Core clustering logic using production-grade algorithms. | |
| """ | |
| import math | |
| # Prepare numeric data | |
| df_clean = df.drop(columns=exclude_columns, errors='ignore') | |
| # Drop datetime columns | |
| datetime_cols = df_clean.select_dtypes(include=['datetime64']).columns.tolist() | |
| df_clean = df_clean.drop(columns=datetime_cols, errors='ignore') | |
| # Get numeric columns only for clustering | |
| numeric_df = df_clean.select_dtypes(include=[np.number]) | |
| if numeric_df.empty or len(numeric_df.columns) < 2: | |
| raise HTTPException(status_code=400, detail="Need at least 2 numeric columns for clustering") | |
| # Fill missing values | |
| numeric_df = numeric_df.fillna(numeric_df.median()) | |
| # Drop zero-variance columns | |
| zero_var_cols = numeric_df.columns[numeric_df.std() == 0].tolist() | |
| numeric_df = numeric_df.drop(columns=zero_var_cols, errors='ignore') | |
| feature_columns = numeric_df.columns.tolist() | |
| X = numeric_df.values | |
| logger.info(f"📊 Clustering {len(X)} samples with {len(feature_columns)} features") | |
| # Scale data | |
| scaler = StandardScaler() | |
| X_scaled = scaler.fit_transform(X) | |
| # Initialize k_scores for elbow method chart | |
| k_scores = None | |
| # ======================================================================= | |
| # AUTO-DETECT OPTIMAL K (if not provided) | |
| # ======================================================================= | |
| if n_clusters is None and algorithm not in ['dbscan']: | |
| n_clusters, k_scores = _find_optimal_k(X_scaled, max_k=min(15, len(X) // 10)) | |
| logger.info(f"✅ Auto-detected optimal k={n_clusters}") | |
| elif n_clusters is None: | |
| n_clusters = 3 # Default for DBSCAN (will auto-detect anyway) | |
| # ======================================================================= | |
| # RUN CLUSTERING (compare all or single algorithm) | |
| # ======================================================================= | |
| if compare_all: | |
| # Run all algorithms and compare | |
| all_results = _compare_all_algorithms(X_scaled, n_clusters) | |
| best_algo = max(all_results, key=lambda x: all_results[x]['silhouette_score']) | |
| best_result = all_results[best_algo] | |
| labels = np.array(best_result['labels']) | |
| algorithm = best_algo | |
| comparison_results = all_results | |
| else: | |
| # Single algorithm | |
| if algorithm == 'auto': | |
| algorithm = 'kmeans' # Default to kmeans for numeric data | |
| labels, model, metrics = _run_single_clustering(X_scaled, algorithm, n_clusters) | |
| comparison_results = None | |
| # ======================================================================= | |
| # CALCULATE METRICS | |
| # ======================================================================= | |
| metrics = _calculate_clustering_metrics(X_scaled, labels) | |
| # Cluster distribution | |
| unique_labels, counts = np.unique(labels, return_counts=True) | |
| cluster_distribution = { | |
| f"Cluster {int(k)}" if k != -1 else "Noise": int(v) | |
| for k, v in zip(unique_labels, counts) | |
| } | |
| actual_n_clusters = len([l for l in unique_labels if l != -1]) | |
| # ======================================================================= | |
| # PCA FOR VISUALIZATION | |
| # ======================================================================= | |
| pca = PCA(n_components=2) | |
| X_2d = pca.fit_transform(X_scaled) | |
| pca_variance = sum(pca.explained_variance_ratio_) | |
| # ======================================================================= | |
| # COMPUTE FEATURE STATISTICS PER CLUSTER | |
| # ======================================================================= | |
| feature_stats = {} | |
| for col in feature_columns: | |
| col_idx = feature_columns.index(col) | |
| stats = { | |
| 'mean': float(numeric_df[col].mean()), | |
| 'std': float(numeric_df[col].std()), | |
| 'min': float(numeric_df[col].min()), | |
| 'max': float(numeric_df[col].max()), | |
| } | |
| feature_stats[col] = stats | |
| # Cluster centroids in original scale | |
| cluster_profiles = {} | |
| for cluster_id in unique_labels: | |
| if cluster_id == -1: | |
| continue | |
| mask = labels == cluster_id | |
| cluster_data = numeric_df.values[mask] | |
| profile = { | |
| 'size': int(mask.sum()), | |
| 'percentage': float(mask.sum() / len(labels) * 100), | |
| 'characteristics': {}, | |
| } | |
| for i, col in enumerate(feature_columns): | |
| profile['characteristics'][col] = { | |
| 'mean': float(np.mean(cluster_data[:, i])), | |
| 'std': float(np.std(cluster_data[:, i])), | |
| } | |
| cluster_profiles[f"Cluster {cluster_id}"] = profile | |
| # ======================================================================= | |
| # GENERATE COMPREHENSIVE CHARTS | |
| # ======================================================================= | |
| charts = _generate_clustering_charts( | |
| X_2d=X_2d, | |
| labels=labels, | |
| algorithm=algorithm, | |
| n_clusters=actual_n_clusters, | |
| silhouette=metrics['silhouette_score'], | |
| feature_names=feature_columns, | |
| cluster_profiles=cluster_profiles, | |
| X_scaled=X_scaled, | |
| X_original=numeric_df.values, | |
| k_scores=k_scores | |
| ) | |
| # ======================================================================= | |
| # RELIABILITY SCORE (Production Intelligence) | |
| # ======================================================================= | |
| reliability_score, validation_warnings = _compute_reliability_score( | |
| n_samples=len(X), | |
| n_features=len(feature_columns), | |
| silhouette=metrics['silhouette_score'], | |
| calinski=metrics.get('calinski_harabasz_score', 0), | |
| davies=metrics.get('davies_bouldin_score', float('inf')), | |
| df=df | |
| ) | |
| # ======================================================================= | |
| # SAVE MODEL FOR PREDICTIONS | |
| # ======================================================================= | |
| model_id = f"clustering_{uuid.uuid4().hex[:8]}" | |
| model_data = { | |
| 'algorithm': algorithm, | |
| 'n_clusters': actual_n_clusters, | |
| 'scaler_mean': scaler.mean_.tolist(), | |
| 'scaler_scale': scaler.scale_.tolist(), | |
| 'feature_columns': feature_columns, | |
| 'labels': labels.tolist(), | |
| 'centroids_scaled': _get_cluster_centroids(X_scaled, labels).tolist() if actual_n_clusters > 0 else [], | |
| 'created_at': datetime.now().isoformat(), | |
| 'silhouette_score': metrics['silhouette_score'], | |
| } | |
| # Save to user's model directory | |
| _save_clustering_model(user_id, model_id, model_data) | |
| # ======================================================================= | |
| # SAVE CLEANED DATA WITH CLUSTER LABELS + PKL MODEL | |
| # ======================================================================= | |
| cleaned_file = None | |
| model_pkl_file = None | |
| # Get user paths (must call here since this is a separate function) | |
| paths = get_user_paths(user_id) | |
| try: | |
| # Create cleaned dataframe with cluster assignments | |
| cleaned_df = df.copy() | |
| cleaned_df['Cluster'] = labels | |
| cleaned_df['Cluster_Name'] = [f'Cluster_{l}' if l >= 0 else 'Noise' for l in labels] | |
| # Add PCA components for visualization | |
| cleaned_df['PCA_1'] = X_2d[:, 0] | |
| cleaned_df['PCA_2'] = X_2d[:, 1] | |
| # Save cleaned CSV | |
| files_dir = paths.get("files", paths["base"] / "files") | |
| files_dir.mkdir(parents=True, exist_ok=True) | |
| cleaned_filename = f"clustered_data_{model_id}.csv" | |
| cleaned_path = files_dir / cleaned_filename | |
| cleaned_df.to_csv(cleaned_path, index=False) | |
| cleaned_file = cleaned_filename | |
| logger.info(f"✅ Saved cleaned data: {cleaned_path}") | |
| # Save PKL model file | |
| import pickle | |
| pkl_filename = f"clustering_model_{model_id}.pkl" | |
| models_dir = paths.get("models", paths["base"] / "models") | |
| models_dir.mkdir(parents=True, exist_ok=True) | |
| pkl_path = models_dir / pkl_filename | |
| pkl_data = { | |
| 'algorithm': algorithm, | |
| 'n_clusters': actual_n_clusters, | |
| 'scaler': scaler, | |
| 'feature_columns': feature_columns, | |
| 'centroids_scaled': _get_cluster_centroids(X_scaled, labels) if actual_n_clusters > 0 else None, | |
| 'labels': labels, | |
| 'model_id': model_id, | |
| 'created_at': datetime.now().isoformat(), | |
| 'silhouette_score': metrics['silhouette_score'], | |
| 'cluster_profiles': cluster_profiles, | |
| } | |
| with open(pkl_path, 'wb') as f: | |
| pickle.dump(pkl_data, f) | |
| model_pkl_file = pkl_filename | |
| logger.info(f"✅ Saved PKL model: {pkl_path}") | |
| except Exception as e: | |
| logger.error(f"❌ Failed to save cleaned data/PKL: {e}") | |
| import traceback | |
| traceback.print_exc() | |
| # ======================================================================= | |
| # SAVE CLUSTERING CHARTS TO DISK (for ZIP download) | |
| # ======================================================================= | |
| if charts: | |
| try: | |
| charts_json_path = paths.get("models", paths["base"] / "models") / "active_clustering_charts.json" | |
| charts_json_path.parent.mkdir(parents=True, exist_ok=True) | |
| with open(charts_json_path, 'w') as f: | |
| json.dump(charts, f) | |
| logger.info(f"✅ Saved {len(charts)} clustering charts to {charts_json_path}") | |
| except Exception as e: | |
| logger.warning(f"Could not save clustering charts: {e}") | |
| # ======================================================================= | |
| # 🔬 PRODUCTION ML ENGINEERING FEATURES | |
| # ======================================================================= | |
| # --- Anomaly Detection (Isolation Forest) --- | |
| anomaly_results = None | |
| try: | |
| from sklearn.ensemble import IsolationForest | |
| iso_forest = IsolationForest(n_estimators=100, contamination=0.05, random_state=42) | |
| anomaly_labels = iso_forest.fit_predict(X_scaled) | |
| anomaly_scores = iso_forest.decision_function(X_scaled) | |
| n_anomalies = int(np.sum(anomaly_labels == -1)) | |
| anomaly_results = { | |
| 'n_anomalies': n_anomalies, | |
| 'anomaly_percentage': round(n_anomalies / len(X_scaled) * 100, 2), | |
| 'anomaly_indices': np.where(anomaly_labels == -1)[0].tolist()[:50], # Limit to 50 | |
| 'anomaly_scores_summary': { | |
| 'min': float(np.min(anomaly_scores)), | |
| 'max': float(np.max(anomaly_scores)), | |
| 'mean': float(np.mean(anomaly_scores)), | |
| 'threshold': float(np.percentile(anomaly_scores, 5)), | |
| } | |
| } | |
| logger.info(f"🔍 Anomaly Detection: {n_anomalies} anomalies ({anomaly_results['anomaly_percentage']}%)") | |
| except Exception as ae: | |
| logger.warning(f"Anomaly detection skipped: {ae}") | |
| # --- t-SNE Visualization (2D) --- | |
| tsne_visualization = None | |
| try: | |
| from sklearn.manifold import TSNE | |
| if len(X_scaled) <= 5000: # t-SNE is expensive for large datasets | |
| perplexity = min(30, max(5, len(X_scaled) // 5)) | |
| tsne = TSNE(n_components=2, perplexity=perplexity, random_state=42, n_iter=500) | |
| X_tsne = tsne.fit_transform(X_scaled) | |
| tsne_visualization = { | |
| 'x': X_tsne[:, 0].tolist(), | |
| 'y': X_tsne[:, 1].tolist(), | |
| } | |
| logger.info(f"📊 t-SNE computed for {len(X_scaled)} samples") | |
| else: | |
| logger.info(f"⏭️ t-SNE skipped: dataset too large ({len(X_scaled)} samples)") | |
| except Exception as te: | |
| logger.warning(f"t-SNE skipped: {te}") | |
| # --- Feature Importance per Cluster --- | |
| feature_importance = None | |
| try: | |
| from sklearn.ensemble import RandomForestClassifier | |
| if actual_n_clusters >= 2 and len(feature_columns) >= 2: | |
| # Use cluster labels as targets, features as input | |
| valid_mask = labels >= 0 # Exclude noise points | |
| if np.sum(valid_mask) > 20: | |
| rf = RandomForestClassifier(n_estimators=50, max_depth=10, random_state=42, n_jobs=-1) | |
| rf.fit(X_scaled[valid_mask], labels[valid_mask]) | |
| importances = rf.feature_importances_ | |
| sorted_idx = np.argsort(importances)[::-1] | |
| feature_importance = { | |
| 'features': [feature_columns[i] for i in sorted_idx], | |
| 'importance_scores': [round(float(importances[i]), 4) for i in sorted_idx], | |
| 'top_3_features': [feature_columns[i] for i in sorted_idx[:3]], | |
| } | |
| logger.info(f"🎯 Feature importance: top features = {feature_importance['top_3_features']}") | |
| except Exception as fie: | |
| logger.warning(f"Feature importance skipped: {fie}") | |
| # --- Cluster Stability (Bootstrap Silhouette Variance) --- | |
| cluster_stability = None | |
| try: | |
| if actual_n_clusters >= 2 and len(X_scaled) >= 50: | |
| n_bootstrap = 5 | |
| bootstrap_scores = [] | |
| for _ in range(n_bootstrap): | |
| sample_idx = np.random.choice(len(X_scaled), size=min(len(X_scaled), 500), replace=True) | |
| X_sample = X_scaled[sample_idx] | |
| try: | |
| km_boot = KMeans(n_clusters=actual_n_clusters, random_state=np.random.randint(1000), n_init=5) | |
| boot_labels = km_boot.fit_predict(X_sample) | |
| if len(set(boot_labels)) > 1: | |
| score = silhouette_score(X_sample, boot_labels) | |
| bootstrap_scores.append(score) | |
| except Exception: | |
| pass | |
| if bootstrap_scores: | |
| cluster_stability = { | |
| 'mean_silhouette': round(float(np.mean(bootstrap_scores)), 4), | |
| 'std_silhouette': round(float(np.std(bootstrap_scores)), 4), | |
| 'stability_rating': 'High' if np.std(bootstrap_scores) < 0.03 else 'Medium' if np.std(bootstrap_scores) < 0.08 else 'Low', | |
| 'n_bootstrap_runs': len(bootstrap_scores), | |
| } | |
| logger.info(f"📈 Cluster stability: {cluster_stability['stability_rating']} (σ={cluster_stability['std_silhouette']})") | |
| except Exception as cse: | |
| logger.warning(f"Cluster stability skipped: {cse}") | |
| # --- Per-Sample Silhouette Scores --- | |
| sample_silhouettes = None | |
| try: | |
| from sklearn.metrics import silhouette_samples | |
| if actual_n_clusters >= 2: | |
| valid_mask = labels >= 0 | |
| if np.sum(valid_mask) > 10 and len(set(labels[valid_mask])) > 1: | |
| sil_samples = silhouette_samples(X_scaled[valid_mask], labels[valid_mask]) | |
| # Per-cluster average silhouette | |
| per_cluster_sil = {} | |
| for cl in range(actual_n_clusters): | |
| mask = labels[valid_mask] == cl | |
| if np.sum(mask) > 0: | |
| per_cluster_sil[f"Cluster {cl}"] = round(float(np.mean(sil_samples[mask])), 4) | |
| sample_silhouettes = { | |
| 'per_cluster': per_cluster_sil, | |
| 'overall_mean': round(float(np.mean(sil_samples)), 4), | |
| 'n_negative': int(np.sum(sil_samples < 0)), # Misclassified samples | |
| 'negative_percentage': round(float(np.sum(sil_samples < 0) / len(sil_samples) * 100), 2), | |
| } | |
| logger.info(f"📊 Silhouette analysis: {sample_silhouettes['n_negative']} potentially misclassified ({sample_silhouettes['negative_percentage']}%)") | |
| except Exception as sse: | |
| logger.warning(f"Sample silhouettes skipped: {sse}") | |
| # ======================================================================= | |
| # BUILD RESPONSE | |
| # ======================================================================= | |
| response = { | |
| 'success': True, | |
| 'model_id': model_id, | |
| 'algorithm': algorithm, | |
| 'n_clusters': actual_n_clusters, | |
| 'n_samples': len(X), | |
| 'n_features': len(feature_columns), | |
| 'feature_columns': feature_columns, | |
| 'feature_stats': feature_stats, | |
| 'silhouette_score': metrics['silhouette_score'], | |
| 'calinski_harabasz_score': metrics.get('calinski_harabasz_score', 0), | |
| 'davies_bouldin_score': metrics.get('davies_bouldin_score', 0), | |
| 'cluster_distribution': cluster_distribution, | |
| 'cluster_profiles': cluster_profiles, | |
| 'pca_variance_explained': pca_variance, | |
| 'labels': labels.tolist(), | |
| 'visualization': { | |
| 'x': X_2d[:, 0].tolist(), | |
| 'y': X_2d[:, 1].tolist(), | |
| }, | |
| 'charts': charts, | |
| 'reliability_score': reliability_score, | |
| 'validation_warnings': validation_warnings if validation_warnings else None, | |
| 'comparison_results': comparison_results, | |
| 'cleaned_file': cleaned_file, | |
| 'model_pkl_file': model_pkl_file, | |
| 'k_scores': k_scores, # For elbow chart on frontend | |
| # 🔬 Production ML Engineering Features | |
| 'anomaly_detection': anomaly_results, | |
| 'tsne_visualization': tsne_visualization, | |
| 'feature_importance': feature_importance, | |
| 'cluster_stability': cluster_stability, | |
| 'sample_silhouettes': sample_silhouettes, | |
| } | |
| logger.info(f"✅ Clustering complete: {actual_n_clusters} clusters, silhouette={metrics['silhouette_score']:.3f}") | |
| return response | |
| def _find_optimal_k(X_scaled: np.ndarray, max_k: int = 10) -> tuple: | |
| """ | |
| Find optimal number of clusters using multiple metrics: | |
| - Silhouette Score (primary) | |
| - Elbow Method (inertia) | |
| - Calinski-Harabasz Index | |
| """ | |
| max_k = min(max_k, len(X_scaled) - 1, 15) | |
| max_k = max(max_k, 3) | |
| scores = {} | |
| inertias = [] | |
| calinski_scores = [] | |
| best_k = 2 | |
| best_score = -1 | |
| for k in range(2, max_k + 1): | |
| try: | |
| # Use k-means++ initialization with multiple runs | |
| kmeans = KMeans( | |
| n_clusters=k, | |
| random_state=42, | |
| n_init=15, # More initializations for stability | |
| max_iter=500, # More iterations for convergence | |
| init='k-means++', | |
| algorithm='lloyd' | |
| ) | |
| labels = kmeans.fit_predict(X_scaled) | |
| if len(set(labels)) > 1: | |
| score = silhouette_score(X_scaled, labels) | |
| scores[k] = score | |
| inertias.append(kmeans.inertia_) | |
| try: | |
| calinski = calinski_harabasz_score(X_scaled, labels) | |
| calinski_scores.append(calinski) | |
| except: | |
| calinski_scores.append(0) | |
| if score > best_score: | |
| best_score = score | |
| best_k = k | |
| except Exception as e: | |
| logger.warning(f"Optimal k search failed for k={k}: {e}") | |
| # If silhouette fails, try elbow method | |
| if best_score < 0 and inertias: | |
| # Find elbow using rate of change | |
| deltas = np.diff(inertias) | |
| if len(deltas) > 1: | |
| delta2 = np.diff(deltas) | |
| elbow_idx = np.argmax(delta2) + 2 | |
| best_k = elbow_idx + 2 # Adjust for range starting at 2 | |
| logger.info(f"🎯 Optimal k detection: best_k={best_k}, silhouette={best_score:.3f}") | |
| return best_k, scores | |
| def _run_single_clustering(X_scaled: np.ndarray, algorithm: str, n_clusters: int): | |
| """ | |
| Run a single clustering algorithm with PRODUCTION-QUALITY settings. | |
| Enhanced with: | |
| - Better hyperparameters | |
| - Multiple initializations | |
| - Adaptive parameters based on data size | |
| """ | |
| metrics = {} | |
| n_samples = len(X_scaled) | |
| if algorithm == 'kmeans': | |
| # Production K-Means with k-means++ and stability settings | |
| model = KMeans( | |
| n_clusters=n_clusters, | |
| random_state=42, | |
| n_init=20, # More initializations for better results | |
| max_iter=500, # More iterations | |
| init='k-means++', # Smart initialization | |
| algorithm='lloyd', | |
| tol=1e-5 # Stricter convergence | |
| ) | |
| labels = model.fit_predict(X_scaled) | |
| metrics['inertia'] = model.inertia_ | |
| metrics['n_iter'] = model.n_iter_ | |
| elif algorithm == 'dbscan': | |
| # Auto-detect eps using k-distance graph with better heuristics | |
| from sklearn.neighbors import NearestNeighbors | |
| # Adaptive k based on data size | |
| k = min(max(5, n_samples // 50), 15, n_samples - 1) | |
| nn = NearestNeighbors(n_neighbors=k) | |
| nn.fit(X_scaled) | |
| distances, _ = nn.kneighbors(X_scaled) | |
| # Use multiple percentiles and pick best | |
| sorted_distances = np.sort(distances[:, -1]) | |
| # Try different eps values and pick best silhouette | |
| best_eps = np.percentile(sorted_distances, 90) | |
| best_labels = None | |
| best_sil = -1 | |
| for pct in [80, 85, 90, 95]: | |
| try: | |
| eps = np.percentile(sorted_distances, pct) | |
| min_samples = max(3, n_samples // 100) | |
| model = DBSCAN(eps=eps, min_samples=min_samples) | |
| test_labels = model.fit_predict(X_scaled) | |
| n_clusters_found = len(set(test_labels)) - (1 if -1 in test_labels else 0) | |
| if n_clusters_found >= 2: | |
| mask = test_labels != -1 | |
| if mask.sum() > 10: | |
| sil = silhouette_score(X_scaled[mask], test_labels[mask]) | |
| if sil > best_sil: | |
| best_sil = sil | |
| best_eps = eps | |
| best_labels = test_labels | |
| except: | |
| pass | |
| if best_labels is not None: | |
| labels = best_labels | |
| else: | |
| model = DBSCAN(eps=best_eps, min_samples=max(3, n_samples // 100)) | |
| labels = model.fit_predict(X_scaled) | |
| metrics['eps'] = best_eps | |
| elif algorithm == 'hierarchical': | |
| # Use ward linkage for compactness, try different linkage if fails | |
| try: | |
| model = AgglomerativeClustering( | |
| n_clusters=n_clusters, | |
| linkage='ward', | |
| metric='euclidean' | |
| ) | |
| labels = model.fit_predict(X_scaled) | |
| except Exception: | |
| # Fallback to average linkage | |
| model = AgglomerativeClustering( | |
| n_clusters=n_clusters, | |
| linkage='average' | |
| ) | |
| labels = model.fit_predict(X_scaled) | |
| metrics['linkage'] = 'ward' | |
| elif algorithm == 'gmm': | |
| # Gaussian Mixture with multiple covariance types and select best | |
| from sklearn.mixture import GaussianMixture | |
| best_model = None | |
| best_bic = float('inf') | |
| for cov_type in ['full', 'tied', 'diag', 'spherical']: | |
| try: | |
| model = GaussianMixture( | |
| n_components=n_clusters, | |
| random_state=42, | |
| n_init=10, | |
| max_iter=200, | |
| covariance_type=cov_type, | |
| init_params='k-means++' | |
| ) | |
| model.fit(X_scaled) | |
| bic = model.bic(X_scaled) | |
| if bic < best_bic: | |
| best_bic = bic | |
| best_model = model | |
| except Exception: | |
| pass | |
| if best_model is None: | |
| # Fallback to simple GMM | |
| best_model = GaussianMixture( | |
| n_components=n_clusters, | |
| random_state=42, | |
| n_init=5 | |
| ) | |
| best_model.fit(X_scaled) | |
| labels = best_model.predict(X_scaled) | |
| metrics['bic'] = best_model.bic(X_scaled) | |
| metrics['aic'] = best_model.aic(X_scaled) | |
| elif algorithm == 'spectral': | |
| # Spectral clustering with adaptive neighbors | |
| n_neighbors = min(max(10, n_samples // 50), 30, n_samples - 1) | |
| try: | |
| model = SpectralClustering( | |
| n_clusters=n_clusters, | |
| random_state=42, | |
| affinity='nearest_neighbors', | |
| n_neighbors=n_neighbors, | |
| assign_labels='cluster_qr' # Better assignment | |
| ) | |
| labels = model.fit_predict(X_scaled) | |
| except Exception: | |
| # Fallback with simpler settings | |
| model = SpectralClustering( | |
| n_clusters=n_clusters, | |
| random_state=42, | |
| affinity='rbf' | |
| ) | |
| labels = model.fit_predict(X_scaled) | |
| else: | |
| raise HTTPException(status_code=400, detail=f"Unknown algorithm: {algorithm}") | |
| return labels, model, metrics | |
| def _compare_all_algorithms(X_scaled: np.ndarray, n_clusters: int) -> Dict[str, Any]: | |
| """Compare all clustering algorithms.""" | |
| results = {} | |
| algorithms = ['kmeans', 'hierarchical', 'gmm', 'spectral', 'dbscan'] | |
| for algo in algorithms: | |
| try: | |
| labels, model, _ = _run_single_clustering(X_scaled, algo, n_clusters) | |
| metrics = _calculate_clustering_metrics(X_scaled, labels) | |
| actual_k = len(set(labels)) - (1 if -1 in labels else 0) | |
| results[algo] = { | |
| 'labels': labels.tolist(), | |
| 'n_clusters': actual_k, | |
| **metrics | |
| } | |
| except Exception as e: | |
| logger.warning(f"Algorithm {algo} failed: {e}") | |
| return results | |
| def _calculate_clustering_metrics(X_scaled: np.ndarray, labels: np.ndarray) -> Dict[str, float]: | |
| """Calculate clustering quality metrics.""" | |
| mask = labels != -1 | |
| unique_labels = set(labels[mask]) | |
| metrics = { | |
| 'silhouette_score': 0.0, | |
| 'calinski_harabasz_score': 0.0, | |
| 'davies_bouldin_score': 0.0, | |
| } | |
| if len(unique_labels) <= 1 or mask.sum() < 2: | |
| return metrics | |
| try: | |
| metrics['silhouette_score'] = float(silhouette_score(X_scaled[mask], labels[mask])) | |
| except Exception: | |
| pass | |
| try: | |
| metrics['calinski_harabasz_score'] = float(calinski_harabasz_score(X_scaled[mask], labels[mask])) | |
| except Exception: | |
| pass | |
| try: | |
| metrics['davies_bouldin_score'] = float(davies_bouldin_score(X_scaled[mask], labels[mask])) | |
| except Exception: | |
| pass | |
| return metrics | |
| def _get_cluster_centroids(X_scaled: np.ndarray, labels: np.ndarray) -> np.ndarray: | |
| """Calculate cluster centroids.""" | |
| unique_labels = [l for l in np.unique(labels) if l != -1] | |
| centroids = [] | |
| for label in unique_labels: | |
| mask = labels == label | |
| centroid = X_scaled[mask].mean(axis=0) | |
| centroids.append(centroid) | |
| return np.array(centroids) if centroids else np.array([]) | |
| def _compute_reliability_score( | |
| n_samples: int, | |
| n_features: int, | |
| silhouette: float, | |
| calinski: float, | |
| davies: float, | |
| df: pd.DataFrame | |
| ) -> tuple: | |
| """Compute reliability score and validation warnings.""" | |
| import math | |
| validation_warnings = [] | |
| # Data quality checks | |
| missing_ratio = df.isna().sum().sum() / df.size if df.size > 0 else 0 | |
| duplicate_ratio = df.duplicated().sum() / n_samples if n_samples > 0 else 0 | |
| if missing_ratio > 0.2: | |
| validation_warnings.append(f"⚠️ High missing data: {missing_ratio:.1%}") | |
| if duplicate_ratio > 0.1: | |
| validation_warnings.append(f"⚠️ Many duplicates: {duplicate_ratio:.1%}") | |
| if n_samples < 100: | |
| validation_warnings.append(f"⚠️ Small dataset ({n_samples} samples)") | |
| if n_features > 50: | |
| validation_warnings.append(f"⚠️ High dimensionality ({n_features} features)") | |
| if silhouette < 0.2: | |
| validation_warnings.append(f"⚠️ Low silhouette score ({silhouette:.3f}) - clusters may overlap") | |
| # Compute reliability score | |
| silhouette_points = max(0, (silhouette + 1) / 2 * 40) | |
| calinski_points = min(25, math.log(calinski + 1) * 3) if calinski > 0 else 0 | |
| davies_points = max(0, 20 - davies * 5) if davies < float('inf') else 10 | |
| if n_samples >= 1000: | |
| size_points = 15 | |
| elif n_samples >= 500: | |
| size_points = 12 | |
| elif n_samples >= 100: | |
| size_points = 8 | |
| else: | |
| size_points = 5 | |
| reliability_score = min(100, silhouette_points + calinski_points + davies_points + size_points) | |
| return reliability_score, validation_warnings | |
| def _generate_clustering_charts( | |
| X_2d: np.ndarray, | |
| labels: np.ndarray, | |
| algorithm: str, | |
| n_clusters: int, | |
| silhouette: float, | |
| feature_names: List[str], | |
| cluster_profiles: Dict, | |
| X_scaled: np.ndarray = None, | |
| X_original: np.ndarray = None, | |
| k_scores: Dict[int, float] = None | |
| ) -> Dict[str, str]: | |
| """ | |
| Generate COMPREHENSIVE visualization charts for unsupervised learning. | |
| Only generates charts when data is sufficient and suitable. | |
| Charts included (when data permits): | |
| 1. Cluster Scatter Plot (PCA 2D) | |
| 2. Elbow Method Graph | |
| 3. Silhouette Score Plot | |
| 4. Cluster Distribution Bar Chart | |
| 5. Cluster Profile Heatmap | |
| 6. PCA Explained Variance Plot | |
| 7. Pairplot (feature relationships) - up to 4 features | |
| 8. 3D Cluster Visualization | |
| 9. Dendrogram (Hierarchical only) | |
| 10. t-SNE Visualization (if samples < 5000) | |
| 11. Boxplot per Cluster | |
| 12. Violin Plot per Cluster | |
| 13. Correlation Heatmap | |
| 14. Radar Chart for Cluster Comparison | |
| """ | |
| charts = {} | |
| try: | |
| import matplotlib | |
| matplotlib.use('Agg') | |
| import matplotlib.pyplot as plt | |
| import seaborn as sns | |
| from matplotlib.colors import LinearSegmentedColormap | |
| # Set style | |
| plt.style.use('seaborn-v0_8-whitegrid') | |
| # Professional color palette | |
| colors = ['#2563eb', '#16a34a', '#dc2626', '#ea580c', '#9333ea', | |
| '#0891b2', '#db2777', '#d97706', '#0d9488', '#4f46e5', | |
| '#84cc16', '#06b6d4', '#f43f5e', '#8b5cf6', '#14b8a6'] | |
| unique_labels = sorted([l for l in set(labels) if l >= 0]) | |
| n_samples = len(labels) | |
| n_features = len(feature_names) if feature_names else 0 | |
| # ================================================================== | |
| # 1. CLUSTER SCATTER PLOT (PCA 2D) - Always generate if possible | |
| # ================================================================== | |
| if X_2d is not None and len(X_2d) > 0: | |
| try: | |
| fig, ax = plt.subplots(figsize=(10, 8)) | |
| for i, label in enumerate(sorted(set(labels))): | |
| mask = labels == label | |
| color = colors[i % len(colors)] if label >= 0 else '#6b7280' | |
| label_name = f'Cluster {label}' if label >= 0 else 'Noise' | |
| ax.scatter(X_2d[mask, 0], X_2d[mask, 1], | |
| c=color, label=label_name, alpha=0.7, s=50, | |
| edgecolors='white', linewidth=0.5) | |
| # Add cluster centroids | |
| for i, label in enumerate(unique_labels): | |
| mask = labels == label | |
| centroid = X_2d[mask].mean(axis=0) | |
| ax.scatter(centroid[0], centroid[1], c=colors[i % len(colors)], | |
| s=200, marker='X', edgecolors='black', linewidth=2, zorder=10) | |
| ax.set_xlabel('PCA Component 1', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('PCA Component 2', fontweight='bold', fontsize=12) | |
| ax.set_title(f'🔮 {algorithm.upper()} Clustering (Silhouette: {silhouette:.3f})', | |
| fontweight='bold', pad=15, fontsize=14) | |
| ax.legend(loc='best', fontsize=10) | |
| ax.grid(alpha=0.3) | |
| fig.tight_layout() | |
| charts['cluster_scatter'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Cluster scatter plot failed: {e}") | |
| # ================================================================== | |
| # 2. ELBOW METHOD GRAPH - Only if k_scores provided | |
| # ================================================================== | |
| if k_scores and len(k_scores) >= 3: | |
| try: | |
| fig, ax = plt.subplots(figsize=(10, 6)) | |
| k_values = sorted(k_scores.keys()) | |
| scores = [k_scores[k] for k in k_values] | |
| ax.plot(k_values, scores, 'bo-', linewidth=2, markersize=10) | |
| # Highlight optimal k | |
| optimal_k = k_values[np.argmax(scores)] | |
| optimal_score = max(scores) | |
| ax.scatter([optimal_k], [optimal_score], c='red', s=200, zorder=10, | |
| marker='*', label=f'Optimal k={optimal_k}') | |
| ax.axvline(x=optimal_k, color='red', linestyle='--', alpha=0.5) | |
| ax.set_xlabel('Number of Clusters (k)', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('Silhouette Score', fontweight='bold', fontsize=12) | |
| ax.set_title('📈 Elbow Method - Optimal Cluster Selection', fontweight='bold', pad=15, fontsize=14) | |
| ax.legend(loc='best') | |
| ax.grid(alpha=0.3) | |
| fig.tight_layout() | |
| charts['elbow_method'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Elbow method chart failed: {e}") | |
| # ================================================================== | |
| # 3. SILHOUETTE SCORE PLOT - Need sklearn and enough samples | |
| # ================================================================== | |
| if X_scaled is not None and n_clusters >= 2 and n_samples >= 50: | |
| try: | |
| from sklearn.metrics import silhouette_samples | |
| fig, ax = plt.subplots(figsize=(10, 8)) | |
| sample_silhouette_values = silhouette_samples(X_scaled, labels) | |
| y_lower = 10 | |
| for i, label in enumerate(unique_labels): | |
| cluster_silhouette_vals = sample_silhouette_values[labels == label] | |
| cluster_silhouette_vals.sort() | |
| cluster_size = cluster_silhouette_vals.shape[0] | |
| y_upper = y_lower + cluster_size | |
| ax.fill_betweenx(np.arange(y_lower, y_upper), | |
| 0, cluster_silhouette_vals, | |
| facecolor=colors[i % len(colors)], alpha=0.7) | |
| ax.text(-0.05, y_lower + 0.5 * cluster_size, f'Cluster {label}', | |
| fontsize=10, fontweight='bold') | |
| y_lower = y_upper + 10 | |
| ax.axvline(x=silhouette, color='red', linestyle='--', linewidth=2, | |
| label=f'Mean: {silhouette:.3f}') | |
| ax.set_xlabel('Silhouette Coefficient', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('Cluster', fontweight='bold', fontsize=12) | |
| ax.set_title('📊 Silhouette Analysis - Cluster Quality', fontweight='bold', pad=15, fontsize=14) | |
| ax.legend(loc='best') | |
| ax.grid(alpha=0.3, axis='x') | |
| fig.tight_layout() | |
| charts['silhouette_plot'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Silhouette plot failed: {e}") | |
| # ================================================================== | |
| # 4. CLUSTER DISTRIBUTION BAR CHART - Always generate | |
| # ================================================================== | |
| try: | |
| fig, ax = plt.subplots(figsize=(10, 6)) | |
| cluster_names = [f'Cluster {l}' if l >= 0 else 'Noise' for l in sorted(set(labels))] | |
| cluster_counts = [np.sum(labels == l) for l in sorted(set(labels))] | |
| cluster_colors = [colors[i % len(colors)] if l >= 0 else '#6b7280' | |
| for i, l in enumerate(sorted(set(labels)))] | |
| bars = ax.bar(cluster_names, cluster_counts, color=cluster_colors, | |
| edgecolor='white', linewidth=2) | |
| # Add counts on bars | |
| for bar, count in zip(bars, cluster_counts): | |
| pct = count / n_samples * 100 | |
| ax.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.5, | |
| f'{count}\n({pct:.1f}%)', ha='center', va='bottom', | |
| fontweight='bold', fontsize=10) | |
| ax.set_xlabel('Cluster', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('Number of Samples', fontweight='bold', fontsize=12) | |
| ax.set_title('📊 Cluster Size Distribution', fontweight='bold', pad=15, fontsize=14) | |
| ax.grid(axis='y', alpha=0.3) | |
| fig.tight_layout() | |
| charts['cluster_distribution'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Distribution chart failed: {e}") | |
| # ================================================================== | |
| # 5. CLUSTER PROFILE HEATMAP - Need profiles and features | |
| # ================================================================== | |
| if cluster_profiles and len(feature_names) >= 3 and len(unique_labels) >= 2: | |
| try: | |
| fig, ax = plt.subplots(figsize=(12, 8)) | |
| # Build heatmap data | |
| top_features = feature_names[:min(15, len(feature_names))] | |
| heatmap_data = [] | |
| cluster_labels_list = [] | |
| for cluster_name, profile in sorted(cluster_profiles.items()): | |
| row = [profile.get(f, {}).get('mean', 0) for f in top_features] | |
| heatmap_data.append(row) | |
| cluster_labels_list.append(cluster_name) | |
| heatmap_array = np.array(heatmap_data) | |
| # Normalize per feature (column) | |
| for j in range(heatmap_array.shape[1]): | |
| col_min = heatmap_array[:, j].min() | |
| col_max = heatmap_array[:, j].max() | |
| if col_max - col_min > 0: | |
| heatmap_array[:, j] = (heatmap_array[:, j] - col_min) / (col_max - col_min) | |
| sns.heatmap(heatmap_array, annot=True, fmt='.2f', cmap='RdYlGn', | |
| xticklabels=top_features, yticklabels=cluster_labels_list, | |
| ax=ax, linewidths=0.5, cbar_kws={'label': 'Normalized Value'}) | |
| ax.set_title('🔥 Cluster Profile Heatmap', fontweight='bold', pad=15, fontsize=14) | |
| ax.set_xlabel('Features', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('Clusters', fontweight='bold', fontsize=12) | |
| plt.xticks(rotation=45, ha='right') | |
| fig.tight_layout() | |
| charts['cluster_heatmap'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Heatmap failed: {e}") | |
| # ================================================================== | |
| # 6. PCA EXPLAINED VARIANCE PLOT - Need enough features | |
| # ================================================================== | |
| if X_scaled is not None and n_features >= 3: | |
| try: | |
| from sklearn.decomposition import PCA as PCAViz | |
| n_components = min(10, n_features, n_samples) | |
| pca_full = PCAViz(n_components=n_components) | |
| pca_full.fit(X_scaled) | |
| fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(14, 5)) | |
| # Individual variance | |
| variance = pca_full.explained_variance_ratio_ | |
| components = range(1, len(variance) + 1) | |
| ax1.bar(components, variance * 100, color=colors[0], edgecolor='white', linewidth=1.5) | |
| ax1.set_xlabel('Principal Component', fontweight='bold', fontsize=12) | |
| ax1.set_ylabel('Variance Explained (%)', fontweight='bold', fontsize=12) | |
| ax1.set_title('📊 Individual Variance per Component', fontweight='bold', pad=15, fontsize=14) | |
| ax1.grid(axis='y', alpha=0.3) | |
| # Cumulative variance | |
| cumulative = np.cumsum(variance) * 100 | |
| ax2.plot(components, cumulative, 'bo-', linewidth=2, markersize=8) | |
| ax2.fill_between(components, cumulative, alpha=0.3, color=colors[0]) | |
| ax2.axhline(y=90, color='red', linestyle='--', label='90% threshold') | |
| ax2.set_xlabel('Number of Components', fontweight='bold', fontsize=12) | |
| ax2.set_ylabel('Cumulative Variance (%)', fontweight='bold', fontsize=12) | |
| ax2.set_title('📈 Cumulative Explained Variance', fontweight='bold', pad=15, fontsize=14) | |
| ax2.legend() | |
| ax2.grid(alpha=0.3) | |
| fig.tight_layout() | |
| charts['pca_variance'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"PCA variance plot failed: {e}") | |
| # ================================================================== | |
| # 7. PAIRPLOT (Feature Relationships) - Need 2-4 features, <2000 samples | |
| # ================================================================== | |
| if X_original is not None and 2 <= n_features <= 6 and n_samples <= 2000: | |
| try: | |
| import pandas as pd | |
| # Select top 4 features | |
| plot_features = feature_names[:min(4, len(feature_names))] | |
| df_plot = pd.DataFrame(X_original[:, :len(plot_features)], columns=plot_features) | |
| df_plot['Cluster'] = [f'C{l}' for l in labels] | |
| palette = {f'C{l}': colors[i % len(colors)] for i, l in enumerate(unique_labels)} | |
| palette['C-1'] = '#6b7280' # Noise | |
| g = sns.pairplot(df_plot, hue='Cluster', palette=palette, | |
| diag_kind='kde', plot_kws={'alpha': 0.6, 's': 30}) | |
| g.fig.suptitle('🔍 Feature Relationships by Cluster', fontweight='bold', y=1.02, fontsize=14) | |
| charts['pairplot'] = _fig_to_base64(g.fig) | |
| plt.close(g.fig) | |
| except Exception as e: | |
| logger.warning(f"Pairplot failed: {e}") | |
| # ================================================================== | |
| # 8. 3D CLUSTER VISUALIZATION - Need 3+ features | |
| # ================================================================== | |
| if X_scaled is not None and n_features >= 3 and n_samples <= 5000: | |
| try: | |
| from sklearn.decomposition import PCA as PCA3D | |
| from mpl_toolkits.mplot3d import Axes3D | |
| pca_3d = PCA3D(n_components=3) | |
| X_3d = pca_3d.fit_transform(X_scaled) | |
| fig = plt.figure(figsize=(12, 10)) | |
| ax = fig.add_subplot(111, projection='3d') | |
| for i, label in enumerate(sorted(set(labels))): | |
| mask = labels == label | |
| color = colors[i % len(colors)] if label >= 0 else '#6b7280' | |
| label_name = f'Cluster {label}' if label >= 0 else 'Noise' | |
| ax.scatter(X_3d[mask, 0], X_3d[mask, 1], X_3d[mask, 2], | |
| c=color, label=label_name, alpha=0.6, s=30) | |
| ax.set_xlabel('PC1', fontweight='bold') | |
| ax.set_ylabel('PC2', fontweight='bold') | |
| ax.set_zlabel('PC3', fontweight='bold') | |
| ax.set_title('🌐 3D Cluster Visualization', fontweight='bold', pad=20, fontsize=14) | |
| ax.legend(loc='best') | |
| charts['cluster_3d'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"3D plot failed: {e}") | |
| # ================================================================== | |
| # 9. DENDROGRAM - For hierarchical clustering (works with any dataset size) | |
| # ================================================================== | |
| if algorithm == 'hierarchical' and X_scaled is not None: | |
| try: | |
| from scipy.cluster.hierarchy import dendrogram, linkage | |
| # Sample data if too large (for performance) | |
| max_dendrogram_samples = min(500, n_samples) | |
| if n_samples > max_dendrogram_samples: | |
| # Random sample for dendrogram visualization | |
| sample_indices = np.random.choice(n_samples, max_dendrogram_samples, replace=False) | |
| X_dendro = X_scaled[sample_indices] | |
| else: | |
| X_dendro = X_scaled | |
| # Use ward linkage for best visual hierarchy | |
| Z = linkage(X_dendro, method='ward') | |
| fig, ax = plt.subplots(figsize=(14, 8)) | |
| # Truncate for readability | |
| dendrogram(Z, ax=ax, truncate_mode='lastp', p=min(50, max_dendrogram_samples), | |
| leaf_rotation=90, leaf_font_size=9, | |
| color_threshold=0.7*max(Z[:,2]), | |
| above_threshold_color='gray') | |
| ax.set_xlabel('Sample Index / Cluster Size', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('Distance (Ward)', fontweight='bold', fontsize=12) | |
| ax.set_title(f'🌳 Hierarchical Clustering Dendrogram ({n_clusters} Clusters)', | |
| fontweight='bold', pad=15, fontsize=14) | |
| ax.axhline(y=0.7*max(Z[:,2]), color='red', linestyle='--', linewidth=2, | |
| label=f'Suggested cut-off for {n_clusters} clusters') | |
| ax.legend(loc='upper right') | |
| ax.grid(axis='y', alpha=0.3) | |
| fig.tight_layout() | |
| charts['dendrogram'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| logger.info(f"✅ Dendrogram generated for hierarchical clustering") | |
| except Exception as e: | |
| logger.warning(f"Dendrogram failed: {e}") | |
| # ================================================================== | |
| # 10. t-SNE VISUALIZATION - Best for <5000 samples | |
| # ================================================================== | |
| if X_scaled is not None and 100 <= n_samples <= 3000 and n_features >= 3: | |
| try: | |
| from sklearn.manifold import TSNE | |
| perplexity = min(30, n_samples // 5) | |
| tsne = TSNE(n_components=2, perplexity=perplexity, random_state=42, n_iter=500) | |
| X_tsne = tsne.fit_transform(X_scaled) | |
| fig, ax = plt.subplots(figsize=(10, 8)) | |
| for i, label in enumerate(sorted(set(labels))): | |
| mask = labels == label | |
| color = colors[i % len(colors)] if label >= 0 else '#6b7280' | |
| label_name = f'Cluster {label}' if label >= 0 else 'Noise' | |
| ax.scatter(X_tsne[mask, 0], X_tsne[mask, 1], | |
| c=color, label=label_name, alpha=0.7, s=40) | |
| ax.set_xlabel('t-SNE 1', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('t-SNE 2', fontweight='bold', fontsize=12) | |
| ax.set_title('🔬 t-SNE Visualization', fontweight='bold', pad=15, fontsize=14) | |
| ax.legend(loc='best') | |
| ax.grid(alpha=0.3) | |
| fig.tight_layout() | |
| charts['tsne'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"t-SNE plot failed: {e}") | |
| # ================================================================== | |
| # 10b. UMAP VISUALIZATION - Optional (requires umap-learn package) | |
| # ================================================================== | |
| if X_scaled is not None and 100 <= n_samples <= 5000 and n_features >= 3: | |
| try: | |
| import umap | |
| reducer = umap.UMAP(n_components=2, random_state=42, n_neighbors=15, min_dist=0.1) | |
| X_umap = reducer.fit_transform(X_scaled) | |
| fig, ax = plt.subplots(figsize=(10, 8)) | |
| for i, label in enumerate(sorted(set(labels))): | |
| mask = labels == label | |
| color = colors[i % len(colors)] if label >= 0 else '#6b7280' | |
| label_name = f'Cluster {label}' if label >= 0 else 'Noise' | |
| ax.scatter(X_umap[mask, 0], X_umap[mask, 1], | |
| c=color, label=label_name, alpha=0.7, s=40) | |
| ax.set_xlabel('UMAP 1', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('UMAP 2', fontweight='bold', fontsize=12) | |
| ax.set_title('🗺️ UMAP Visualization', fontweight='bold', pad=15, fontsize=14) | |
| ax.legend(loc='best') | |
| ax.grid(alpha=0.3) | |
| fig.tight_layout() | |
| charts['umap'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except ImportError: | |
| logger.info("UMAP not installed - skipping UMAP visualization") | |
| except Exception as e: | |
| logger.warning(f"UMAP plot failed: {e}") | |
| # ================================================================== | |
| # 11. BOXPLOT PER CLUSTER - Need features | |
| # ================================================================== | |
| if X_original is not None and n_features >= 2 and len(unique_labels) >= 2: | |
| try: | |
| import pandas as pd | |
| # Select top 4 features for box plots | |
| top_features = feature_names[:min(4, len(feature_names))] | |
| n_plot = len(top_features) | |
| fig, axes = plt.subplots(1, n_plot, figsize=(4*n_plot, 6)) | |
| if n_plot == 1: | |
| axes = [axes] | |
| for idx, feature in enumerate(top_features): | |
| feature_idx = feature_names.index(feature) | |
| data_by_cluster = [X_original[labels == l, feature_idx] for l in unique_labels] | |
| bp = axes[idx].boxplot(data_by_cluster, patch_artist=True, | |
| labels=[f'C{l}' for l in unique_labels]) | |
| for i, patch in enumerate(bp['boxes']): | |
| patch.set_facecolor(colors[i % len(colors)]) | |
| patch.set_alpha(0.7) | |
| axes[idx].set_xlabel('Cluster', fontweight='bold', fontsize=11) | |
| axes[idx].set_ylabel(feature, fontweight='bold', fontsize=11) | |
| axes[idx].set_title(f'{feature}', fontweight='bold', fontsize=12) | |
| axes[idx].grid(axis='y', alpha=0.3) | |
| fig.suptitle('📦 Feature Distribution by Cluster (Boxplots)', fontweight='bold', fontsize=14, y=1.02) | |
| fig.tight_layout() | |
| charts['boxplots'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Boxplot failed: {e}") | |
| # ================================================================== | |
| # 12. VIOLIN PLOT PER CLUSTER - Alternative to boxplot | |
| # ================================================================== | |
| if X_original is not None and n_features >= 2 and len(unique_labels) >= 2 and n_samples >= 50: | |
| try: | |
| import pandas as pd | |
| # Select top 3 features for violin plots | |
| top_features = feature_names[:min(3, len(feature_names))] | |
| fig, axes = plt.subplots(1, len(top_features), figsize=(5*len(top_features), 6)) | |
| if len(top_features) == 1: | |
| axes = [axes] | |
| for idx, feature in enumerate(top_features): | |
| feature_idx = feature_names.index(feature) | |
| # Create dataframe for seaborn | |
| df_violin = pd.DataFrame({ | |
| 'Value': X_original[:, feature_idx], | |
| 'Cluster': [f'C{l}' for l in labels] | |
| }) | |
| palette = {f'C{l}': colors[i % len(colors)] for i, l in enumerate(unique_labels)} | |
| sns.violinplot(data=df_violin, x='Cluster', y='Value', | |
| palette=palette, ax=axes[idx]) | |
| axes[idx].set_xlabel('Cluster', fontweight='bold', fontsize=11) | |
| axes[idx].set_ylabel(feature, fontweight='bold', fontsize=11) | |
| axes[idx].set_title(f'{feature}', fontweight='bold', fontsize=12) | |
| axes[idx].grid(axis='y', alpha=0.3) | |
| fig.suptitle('🎻 Feature Distribution by Cluster (Violin Plots)', fontweight='bold', fontsize=14, y=1.02) | |
| fig.tight_layout() | |
| charts['violin_plots'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Violin plot failed: {e}") | |
| # ================================================================== | |
| # 13. CORRELATION HEATMAP - Need enough features | |
| # ================================================================== | |
| if X_original is not None and n_features >= 3: | |
| try: | |
| import pandas as pd | |
| df_corr = pd.DataFrame(X_original, columns=feature_names) | |
| corr_matrix = df_corr.corr() | |
| fig, ax = plt.subplots(figsize=(10, 8)) | |
| mask = np.triu(np.ones_like(corr_matrix, dtype=bool)) | |
| sns.heatmap(corr_matrix, mask=mask, annot=True, fmt='.2f', | |
| cmap='coolwarm', center=0, ax=ax, | |
| linewidths=0.5, cbar_kws={'label': 'Correlation'}) | |
| ax.set_title('🔗 Feature Correlation Heatmap', fontweight='bold', pad=15, fontsize=14) | |
| plt.xticks(rotation=45, ha='right') | |
| fig.tight_layout() | |
| charts['correlation_heatmap'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Correlation heatmap failed: {e}") | |
| # ================================================================== | |
| # 14. RADAR CHART FOR CLUSTER COMPARISON - Need profiles | |
| # ================================================================== | |
| if cluster_profiles and len(feature_names) >= 3 and len(unique_labels) >= 2: | |
| try: | |
| fig, ax = plt.subplots(figsize=(10, 10), subplot_kw=dict(polar=True)) | |
| top_features = feature_names[:min(8, len(feature_names))] | |
| n_features_plot = len(top_features) | |
| angles = np.linspace(0, 2 * np.pi, n_features_plot, endpoint=False).tolist() | |
| angles += angles[:1] | |
| for cluster_name, profile in sorted(cluster_profiles.items()): | |
| values = [profile.get(f, {}).get('mean', 0) for f in top_features] | |
| # Normalize to 0-1 range across all clusters | |
| all_vals = [cluster_profiles[cn].get(f, {}).get('mean', 0) | |
| for cn in cluster_profiles for f in top_features] | |
| val_min, val_max = min(all_vals), max(all_vals) | |
| if val_max - val_min > 0: | |
| values = [(v - val_min) / (val_max - val_min) for v in values] | |
| values += values[:1] | |
| cluster_idx = int(cluster_name.split()[-1]) | |
| ax.plot(angles, values, 'o-', linewidth=2, | |
| label=cluster_name, color=colors[cluster_idx % len(colors)]) | |
| ax.fill(angles, values, alpha=0.15, color=colors[cluster_idx % len(colors)]) | |
| ax.set_xticks(angles[:-1]) | |
| ax.set_xticklabels(top_features, fontsize=10) | |
| ax.set_title('🎯 Radar Chart - Cluster Comparison', fontweight='bold', pad=25, fontsize=14) | |
| ax.legend(loc='upper right', bbox_to_anchor=(1.3, 1.1)) | |
| fig.tight_layout() | |
| charts['radar_chart'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| except Exception as e: | |
| logger.warning(f"Radar chart failed: {e}") | |
| # ================================================================== | |
| # 15. GMM BIC/AIC SCORES - Only for GMM algorithm | |
| # ================================================================== | |
| if algorithm == 'gmm' and X_scaled is not None: | |
| try: | |
| from sklearn.mixture import GaussianMixture | |
| k_range = range(2, min(11, n_samples // 10 + 1)) | |
| bic_scores = [] | |
| aic_scores = [] | |
| for k in k_range: | |
| gmm = GaussianMixture(n_components=k, random_state=42, n_init=3) | |
| gmm.fit(X_scaled) | |
| bic_scores.append(gmm.bic(X_scaled)) | |
| aic_scores.append(gmm.aic(X_scaled)) | |
| fig, ax = plt.subplots(figsize=(10, 6)) | |
| ax.plot(list(k_range), bic_scores, 'bo-', linewidth=2, markersize=8, label='BIC') | |
| ax.plot(list(k_range), aic_scores, 'rs-', linewidth=2, markersize=8, label='AIC') | |
| # Highlight optimal k (lowest BIC) | |
| optimal_k = list(k_range)[np.argmin(bic_scores)] | |
| ax.axvline(x=optimal_k, color='green', linestyle='--', alpha=0.7, | |
| label=f'Optimal k={optimal_k}') | |
| ax.scatter([optimal_k], [min(bic_scores)], c='green', s=200, zorder=10, marker='*') | |
| ax.set_xlabel('Number of Components (k)', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('Score (lower is better)', fontweight='bold', fontsize=12) | |
| ax.set_title('📊 GMM Model Selection - BIC/AIC Scores', fontweight='bold', pad=15, fontsize=14) | |
| ax.legend(loc='best') | |
| ax.grid(alpha=0.3) | |
| fig.tight_layout() | |
| charts['gmm_bic_aic'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| logger.info(f"✅ GMM BIC/AIC chart generated") | |
| except Exception as e: | |
| logger.warning(f"GMM BIC/AIC chart failed: {e}") | |
| # ================================================================== | |
| # 16. DBSCAN K-DISTANCE PLOT - Only for DBSCAN algorithm | |
| # ================================================================== | |
| if algorithm == 'dbscan' and X_scaled is not None and n_samples >= 20: | |
| try: | |
| from sklearn.neighbors import NearestNeighbors | |
| # Calculate k-distance (k=min_samples typically 5 or n/100) | |
| k = max(5, min(n_samples // 100, 20)) | |
| nbrs = NearestNeighbors(n_neighbors=k).fit(X_scaled) | |
| distances, _ = nbrs.kneighbors(X_scaled) | |
| k_distances = np.sort(distances[:, k-1])[::-1] | |
| fig, ax = plt.subplots(figsize=(10, 6)) | |
| ax.plot(range(1, len(k_distances) + 1), k_distances, 'b-', linewidth=2) | |
| # Find elbow point (approximate) | |
| kneedle_idx = len(k_distances) // 4 | |
| if kneedle_idx > 0: | |
| suggested_eps = k_distances[kneedle_idx] | |
| ax.axhline(y=suggested_eps, color='red', linestyle='--', | |
| label=f'Suggested eps ≈ {suggested_eps:.3f}') | |
| ax.scatter([kneedle_idx], [suggested_eps], c='red', s=100, zorder=10) | |
| ax.set_xlabel('Points (sorted by distance)', fontweight='bold', fontsize=12) | |
| ax.set_ylabel(f'{k}-th Nearest Neighbor Distance', fontweight='bold', fontsize=12) | |
| ax.set_title('📈 DBSCAN Epsilon Selection (K-Distance Graph)', fontweight='bold', pad=15, fontsize=14) | |
| ax.legend(loc='best') | |
| ax.grid(alpha=0.3) | |
| fig.tight_layout() | |
| charts['dbscan_kdist'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| logger.info(f"✅ DBSCAN k-distance chart generated") | |
| except Exception as e: | |
| logger.warning(f"DBSCAN k-distance chart failed: {e}") | |
| # ================================================================== | |
| # 17. SPECTRAL AFFINITY MATRIX - Only for Spectral algorithm | |
| # ================================================================== | |
| if algorithm == 'spectral' and X_scaled is not None and n_samples <= 500: | |
| try: | |
| from sklearn.metrics import pairwise_distances | |
| # Sample if too large | |
| sample_size = min(200, n_samples) | |
| if n_samples > sample_size: | |
| indices = np.random.choice(n_samples, sample_size, replace=False) | |
| X_sample = X_scaled[indices] | |
| labels_sample = labels[indices] | |
| else: | |
| X_sample = X_scaled | |
| labels_sample = labels | |
| # Compute affinity matrix (RBF kernel) | |
| gamma = 1.0 / X_sample.shape[1] | |
| distances = pairwise_distances(X_sample) | |
| affinity = np.exp(-gamma * distances ** 2) | |
| # Sort by cluster labels for better visualization | |
| sorted_indices = np.argsort(labels_sample) | |
| affinity_sorted = affinity[sorted_indices][:, sorted_indices] | |
| fig, ax = plt.subplots(figsize=(10, 8)) | |
| sns.heatmap(affinity_sorted, cmap='viridis', ax=ax, | |
| xticklabels=False, yticklabels=False, | |
| cbar_kws={'label': 'Affinity (similarity)'}) | |
| ax.set_title('🔗 Spectral Clustering - Affinity Matrix', fontweight='bold', pad=15, fontsize=14) | |
| ax.set_xlabel('Samples (sorted by cluster)', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('Samples (sorted by cluster)', fontweight='bold', fontsize=12) | |
| fig.tight_layout() | |
| charts['spectral_affinity'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| logger.info(f"✅ Spectral affinity matrix chart generated") | |
| except Exception as e: | |
| logger.warning(f"Spectral affinity chart failed: {e}") | |
| # ================================================================== | |
| # 19. FEATURE IMPORTANCE - Variance-based importance for clustering | |
| # ================================================================== | |
| if X_scaled is not None and feature_names and n_clusters >= 2: | |
| try: | |
| fig, ax = plt.subplots(figsize=(10, 6)) | |
| # Compute feature importance using between-cluster variance ratio | |
| importances = [] | |
| for j in range(X_scaled.shape[1]): | |
| overall_mean = X_scaled[:, j].mean() | |
| between_var = sum( | |
| np.sum(labels == label) * (X_scaled[labels == label, j].mean() - overall_mean) ** 2 | |
| for label in unique_labels | |
| ) | |
| total_var = np.var(X_scaled[:, j]) * len(X_scaled) | |
| importances.append(between_var / max(total_var, 1e-10)) | |
| importances = np.array(importances) | |
| sorted_idx = np.argsort(importances)[::-1] | |
| top_n = min(15, len(feature_names)) | |
| top_idx = sorted_idx[:top_n] | |
| feat_labels = [feature_names[i] if i < len(feature_names) else f"F{i}" for i in top_idx] | |
| feat_values = importances[top_idx] | |
| bars = ax.barh(range(top_n), feat_values[::-1], color=plt.cm.viridis(np.linspace(0.3, 0.9, top_n))) | |
| ax.set_yticks(range(top_n)) | |
| ax.set_yticklabels(feat_labels[::-1], fontsize=10) | |
| ax.set_xlabel('Cluster Separation Importance', fontweight='bold', fontsize=12) | |
| ax.set_title('⭐ Feature Importance for Cluster Separation', fontweight='bold', pad=15, fontsize=14) | |
| ax.grid(alpha=0.3, axis='x') | |
| fig.tight_layout() | |
| charts['feature_importance'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| logger.info("✅ Feature importance chart generated") | |
| except Exception as e: | |
| logger.warning(f"Feature importance chart failed: {e}") | |
| # ================================================================== | |
| # 20. SILHOUETTE COMPARISON - Compare silhouette across k values | |
| # ================================================================== | |
| if k_scores and len(k_scores) >= 3: | |
| try: | |
| fig, ax = plt.subplots(figsize=(10, 6)) | |
| k_vals = sorted(k_scores.keys()) | |
| s_scores = [k_scores[k] for k in k_vals] | |
| colors_bar = ['#ef4444' if s < 0.25 else '#f59e0b' if s < 0.5 else '#22c55e' if s < 0.7 else '#3b82f6' for s in s_scores] | |
| bars = ax.bar(k_vals, s_scores, color=colors_bar, edgecolor='white', linewidth=1.5, width=0.6) | |
| # Add value labels | |
| for bar, score in zip(bars, s_scores): | |
| ax.text(bar.get_x() + bar.get_width() / 2., bar.get_height() + 0.01, | |
| f'{score:.3f}', ha='center', va='bottom', fontweight='bold', fontsize=10) | |
| # Add quality zones | |
| ax.axhline(y=0.25, color='#ef4444', linestyle='--', alpha=0.4, label='Poor (<0.25)') | |
| ax.axhline(y=0.5, color='#f59e0b', linestyle='--', alpha=0.4, label='Fair (0.25-0.5)') | |
| ax.axhline(y=0.7, color='#22c55e', linestyle='--', alpha=0.4, label='Good (0.5-0.7)') | |
| ax.set_xlabel('Number of Clusters (k)', fontweight='bold', fontsize=12) | |
| ax.set_ylabel('Silhouette Score', fontweight='bold', fontsize=12) | |
| ax.set_title('📊 Silhouette Score Comparison Across k', fontweight='bold', pad=15, fontsize=14) | |
| ax.set_xticks(k_vals) | |
| ax.legend(loc='best', fontsize=9) | |
| ax.set_ylim(0, max(s_scores) * 1.2 if s_scores else 1) | |
| ax.grid(alpha=0.3, axis='y') | |
| fig.tight_layout() | |
| charts['silhouette_comparison'] = _fig_to_base64(fig) | |
| plt.close(fig) | |
| logger.info("✅ Silhouette comparison chart generated") | |
| except Exception as e: | |
| logger.warning(f"Silhouette comparison chart failed: {e}") | |
| logger.info(f"✅ Generated {len(charts)} clustering charts") | |
| except Exception as e: | |
| logger.error(f"Chart generation error: {e}") | |
| import traceback | |
| traceback.print_exc() | |
| return charts | |
| def _fig_to_base64(fig) -> str: | |
| """Convert matplotlib figure to base64 string.""" | |
| buf = io.BytesIO() | |
| fig.savefig(buf, format='png', dpi=150, bbox_inches='tight', facecolor='white') | |
| buf.seek(0) | |
| return f"data:image/png;base64,{base64.b64encode(buf.getvalue()).decode('utf-8')}" | |
| def _save_clustering_model(user_id: str, model_id: str, model_data: Dict): | |
| """Save clustering model for predictions.""" | |
| try: | |
| paths = get_user_paths(user_id) | |
| models_dir = paths.get("models", paths["base"] / "models") | |
| models_dir.mkdir(parents=True, exist_ok=True) | |
| model_path = models_dir / f"{model_id}.json" | |
| with open(model_path, 'w') as f: | |
| json.dump(model_data, f) | |
| # Also save as active clustering model | |
| active_path = models_dir / "active_clustering.json" | |
| with open(active_path, 'w') as f: | |
| json.dump({'model_id': model_id, **model_data}, f) | |
| logger.info(f"✅ Saved clustering model: {model_path}") | |
| except Exception as e: | |
| logger.warning(f"Failed to save model: {e}") | |
| # =========================================================================== | |
| # 🔮 CLUSTER PREDICTION ENDPOINT | |
| # =========================================================================== | |
| async def predict_cluster( | |
| request: ClusterPredictRequest, | |
| x_user_id: Optional[str] = Header(None, alias="X-User-ID") | |
| ): | |
| """ | |
| Predict which cluster a new data point belongs to. | |
| Uses the saved model to find the nearest cluster centroid. | |
| """ | |
| user_id = request.user_id or x_user_id | |
| if not user_id: | |
| raise HTTPException(status_code=400, detail="User ID required") | |
| try: | |
| # Load model | |
| paths = get_user_paths(user_id) | |
| models_dir = paths.get("models", paths["base"] / "models") | |
| model_path = models_dir / f"{request.model_id}.json" | |
| if not model_path.exists(): | |
| # Try active model | |
| model_path = models_dir / "active_clustering.json" | |
| if not model_path.exists(): | |
| raise HTTPException(status_code=404, detail="Clustering model not found. Run clustering first.") | |
| with open(model_path, 'r') as f: | |
| model_data = json.load(f) | |
| # Prepare input features | |
| feature_columns = model_data['feature_columns'] | |
| features_array = np.array([request.features.get(col, 0) for col in feature_columns]).reshape(1, -1) | |
| # Scale features | |
| scaler_mean = np.array(model_data['scaler_mean']) | |
| scaler_scale = np.array(model_data['scaler_scale']) | |
| features_scaled = (features_array - scaler_mean) / scaler_scale | |
| # Find nearest centroid | |
| centroids = np.array(model_data['centroids_scaled']) | |
| if len(centroids) == 0: | |
| raise HTTPException(status_code=400, detail="No cluster centroids available") | |
| distances = np.linalg.norm(centroids - features_scaled, axis=1) | |
| predicted_cluster = int(np.argmin(distances)) | |
| confidence = float(1.0 / (1.0 + distances[predicted_cluster])) | |
| # Get cluster characteristics | |
| cluster_name = f"Cluster {predicted_cluster}" | |
| return { | |
| 'success': True, | |
| 'cluster': predicted_cluster, | |
| 'cluster_name': cluster_name, | |
| 'confidence': confidence, | |
| 'distances_to_centroids': distances.tolist(), | |
| 'algorithm': model_data.get('algorithm', 'unknown'), | |
| 'n_clusters': model_data.get('n_clusters', len(centroids)), | |
| } | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| logger.error(f"Cluster prediction failed: {e}") | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| async def get_active_clustering_model(user_id: str): | |
| """Get the active clustering model for a user.""" | |
| try: | |
| paths = get_user_paths(user_id) | |
| models_dir = paths.get("models", paths["base"] / "models") | |
| active_path = models_dir / "active_clustering.json" | |
| if not active_path.exists(): | |
| return {'success': True, 'has_model': False} | |
| with open(active_path, 'r') as f: | |
| model_data = json.load(f) | |
| return { | |
| 'success': True, | |
| 'has_model': True, | |
| 'model_id': model_data.get('model_id'), | |
| 'algorithm': model_data.get('algorithm'), | |
| 'n_clusters': model_data.get('n_clusters'), | |
| 'created_at': model_data.get('created_at'), | |
| } | |
| except Exception as e: | |
| logger.error(f"Failed to get active model: {e}") | |
| return {'success': False, 'error': str(e)} | |
| # =========================================================================== | |
| # LEGACY FILE UPLOAD ENDPOINTS (Keep for backward compatibility) | |
| # =========================================================================== | |
| async def perform_clustering( | |
| file: UploadFile = File(...), | |
| algorithm: str = Form("kmeans"), | |
| n_clusters: Optional[int] = Form(3), | |
| features: Optional[str] = Form(None), | |
| eps: Optional[float] = Form(0.5), | |
| min_samples: Optional[int] = Form(5), | |
| normalize: bool = Form(True) | |
| ): | |
| """ | |
| Perform clustering analysis on uploaded data | |
| Parameters: | |
| - algorithm: 'kmeans', 'dbscan', 'gmm', 'spectral' | |
| - n_clusters: Number of clusters (for kmeans, gmm, spectral) | |
| - features: Comma-separated feature names (optional) | |
| - eps: DBSCAN epsilon parameter | |
| - min_samples: DBSCAN min samples parameter | |
| - normalize: Whether to normalize features | |
| """ | |
| try: | |
| # Read uploaded file | |
| contents = await file.read() | |
| df = pd.read_csv(io.BytesIO(contents)) | |
| logger.info(f"📊 Clustering {len(df)} rows with {algorithm.upper()}") | |
| # Select features | |
| if features: | |
| feature_list = [f.strip() for f in features.split(",")] | |
| X = df[feature_list].select_dtypes(include=[np.number]) | |
| else: | |
| X = df.select_dtypes(include=[np.number]) | |
| if X.empty: | |
| raise HTTPException(status_code=400, detail="No numeric features found") | |
| # Handle missing values | |
| X = X.fillna(X.mean()) | |
| # Normalize if requested | |
| if normalize: | |
| scaler = StandardScaler() | |
| X_scaled = scaler.fit_transform(X) | |
| else: | |
| X_scaled = X.values | |
| # Perform clustering | |
| if algorithm == "kmeans": | |
| model = KMeans(n_clusters=n_clusters, random_state=42, n_init=10) | |
| labels = model.fit_predict(X_scaled) | |
| cluster_centers = model.cluster_centers_ | |
| elif algorithm == "dbscan": | |
| model = DBSCAN(eps=eps, min_samples=min_samples) | |
| labels = model.fit_predict(X_scaled) | |
| cluster_centers = None | |
| n_clusters = len(set(labels)) - (1 if -1 in labels else 0) | |
| elif algorithm == "gmm": | |
| model = GaussianMixture(n_components=n_clusters, random_state=42) | |
| labels = model.fit_predict(X_scaled) | |
| cluster_centers = model.means_ | |
| elif algorithm == "spectral": | |
| model = SpectralClustering(n_clusters=n_clusters, random_state=42) | |
| labels = model.fit_predict(X_scaled) | |
| cluster_centers = None | |
| else: | |
| raise HTTPException(status_code=400, detail=f"Unknown algorithm: {algorithm}") | |
| # Calculate metrics (only if more than 1 cluster) | |
| metrics = {} | |
| if len(set(labels)) > 1 and len(set(labels)) < len(X_scaled): | |
| try: | |
| metrics["silhouette_score"] = float(silhouette_score(X_scaled, labels)) | |
| metrics["davies_bouldin_score"] = float(davies_bouldin_score(X_scaled, labels)) | |
| metrics["calinski_harabasz_score"] = float(calinski_harabasz_score(X_scaled, labels)) | |
| except Exception as e: | |
| logger.warning(f"Could not calculate metrics: {e}") | |
| # PCA for visualization (2D) | |
| pca = PCA(n_components=2) | |
| X_pca = pca.fit_transform(X_scaled) | |
| # Prepare results | |
| cluster_counts = pd.Series(labels).value_counts().to_dict() | |
| # ============================================================= | |
| # 🛡️ PRODUCTION INTELLIGENCE: Data quality & reliability | |
| # ============================================================= | |
| reliability_score = 75 # Default | |
| validation_warnings = [] | |
| data_quality = {} | |
| try: | |
| # 1. Check data quality | |
| n_samples = len(df) | |
| n_features = X.shape[1] | |
| missing_ratio = df.isna().sum().sum() / df.size if df.size > 0 else 0 | |
| duplicate_ratio = df.duplicated().sum() / n_samples if n_samples > 0 else 0 | |
| data_quality = { | |
| 'n_samples': n_samples, | |
| 'n_features': n_features, | |
| 'missing_ratio': float(missing_ratio), | |
| 'duplicate_ratio': float(duplicate_ratio), | |
| 'size_category': 'small' if n_samples < 500 else 'medium' if n_samples < 5000 else 'large' | |
| } | |
| # Data quality warnings | |
| if missing_ratio > 0.2: | |
| validation_warnings.append(f"⚠️ High missing data: {missing_ratio:.1%} - may affect cluster quality") | |
| if duplicate_ratio > 0.1: | |
| validation_warnings.append(f"⚠️ Many duplicates: {duplicate_ratio:.1%} - consider removing") | |
| if n_samples < 100: | |
| validation_warnings.append(f"⚠️ Small dataset ({n_samples} samples) - clusters may be unreliable") | |
| if n_features > 50: | |
| validation_warnings.append(f"⚠️ High dimensionality ({n_features} features) - consider dimensionality reduction") | |
| # 2. Compute reliability score for clustering | |
| # Based on silhouette score, cluster separation, and sample size | |
| silhouette = metrics.get('silhouette_score', 0) | |
| calinski = metrics.get('calinski_harabasz_score', 0) | |
| davies = metrics.get('davies_bouldin_score', float('inf')) | |
| # Silhouette contributes 40 points (scaled from -1 to 1) | |
| silhouette_points = max(0, (silhouette + 1) / 2 * 40) | |
| # Calinski-Harabasz contributes 25 points (log-scaled) | |
| import math | |
| calinski_points = min(25, math.log(calinski + 1) * 3) if calinski > 0 else 0 | |
| # Davies-Bouldin contributes 20 points (lower is better) | |
| davies_points = max(0, 20 - davies * 5) if davies < float('inf') else 10 | |
| # Sample size contributes 15 points | |
| if n_samples >= 1000: | |
| size_points = 15 | |
| elif n_samples >= 500: | |
| size_points = 12 | |
| elif n_samples >= 100: | |
| size_points = 8 | |
| else: | |
| size_points = 5 | |
| reliability_score = min(100, silhouette_points + calinski_points + davies_points + size_points) | |
| # Add cluster quality assessment | |
| if silhouette < 0.2: | |
| validation_warnings.append(f"⚠️ Low silhouette score ({silhouette:.3f}) - clusters may overlap significantly") | |
| if silhouette > 0.7: | |
| logger.info(f"✅ Excellent cluster separation (silhouette={silhouette:.3f})") | |
| logger.info(f"🛡️ Clustering Reliability Score: {reliability_score:.1f}/100") | |
| except Exception as intel_err: | |
| logger.warning(f"Production Intelligence check failed: {intel_err}") | |
| result = { | |
| "algorithm": algorithm, | |
| "n_clusters": n_clusters if algorithm != "dbscan" else len(set(labels)) - (1 if -1 in labels else 0), | |
| "n_samples": len(df), | |
| "n_features": X.shape[1], | |
| "features_used": list(X.columns), | |
| "labels": labels.tolist(), | |
| "cluster_counts": {str(k): int(v) for k, v in cluster_counts.items()}, | |
| "metrics": metrics, | |
| "visualization": { | |
| "x": X_pca[:, 0].tolist(), | |
| "y": X_pca[:, 1].tolist(), | |
| "explained_variance": pca.explained_variance_ratio_.tolist() | |
| }, | |
| # 🛡️ PRODUCTION INTELLIGENCE outputs | |
| "reliability_score": reliability_score, | |
| "validation_warnings": validation_warnings if validation_warnings else None, | |
| "data_quality": data_quality, | |
| } | |
| # Add cluster centers if available | |
| if cluster_centers is not None: | |
| if normalize: | |
| cluster_centers = scaler.inverse_transform(cluster_centers) | |
| result["cluster_centers"] = cluster_centers.tolist() | |
| logger.info(f"✅ Clustering complete: {result['n_clusters']} clusters found") | |
| return result | |
| except pd.errors.EmptyDataError: | |
| raise HTTPException(status_code=400, detail="Empty CSV file") | |
| except KeyError as e: | |
| raise HTTPException(status_code=400, detail=f"Feature not found: {str(e)}") | |
| except Exception as e: | |
| logger.error(f"❌ Clustering error: {e}") | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| async def find_optimal_clusters( | |
| file: UploadFile = File(...), | |
| algorithm: str = Form("kmeans"), | |
| max_clusters: int = Form(10), | |
| features: Optional[str] = Form(None), | |
| normalize: bool = Form(True) | |
| ): | |
| """ | |
| Find optimal number of clusters using elbow method and silhouette analysis | |
| """ | |
| try: | |
| # Read uploaded file | |
| contents = await file.read() | |
| df = pd.read_csv(io.BytesIO(contents)) | |
| # Select features | |
| if features: | |
| feature_list = [f.strip() for f in features.split(",")] | |
| X = df[feature_list].select_dtypes(include=[np.number]) | |
| else: | |
| X = df.select_dtypes(include=[np.number]) | |
| if X.empty: | |
| raise HTTPException(status_code=400, detail="No numeric features found") | |
| # Handle missing values | |
| X = X.fillna(X.mean()) | |
| # Normalize if requested | |
| if normalize: | |
| scaler = StandardScaler() | |
| X_scaled = scaler.fit_transform(X) | |
| else: | |
| X_scaled = X.values | |
| # Test different numbers of clusters | |
| inertias = [] | |
| silhouettes = [] | |
| k_range = range(2, min(max_clusters + 1, len(X_scaled))) | |
| for k in k_range: | |
| if algorithm == "kmeans": | |
| model = KMeans(n_clusters=k, random_state=42, n_init=10) | |
| labels = model.fit_predict(X_scaled) | |
| inertias.append(model.inertia_) | |
| elif algorithm == "gmm": | |
| model = GaussianMixture(n_components=k, random_state=42) | |
| labels = model.fit_predict(X_scaled) | |
| inertias.append(-model.bic(X_scaled)) # Negative BIC | |
| else: | |
| raise HTTPException(status_code=400, detail=f"Optimal search not supported for {algorithm}") | |
| # Calculate silhouette score | |
| silhouette = silhouette_score(X_scaled, labels) | |
| silhouettes.append(silhouette) | |
| # Find optimal k (highest silhouette) | |
| optimal_k = list(k_range)[np.argmax(silhouettes)] | |
| return { | |
| "algorithm": algorithm, | |
| "k_range": list(k_range), | |
| "inertias": inertias, | |
| "silhouette_scores": silhouettes, | |
| "optimal_k": optimal_k, | |
| "max_silhouette": max(silhouettes) | |
| } | |
| except Exception as e: | |
| logger.error(f"❌ Optimal clustering error: {e}") | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| async def get_clustering_info(): | |
| """ | |
| Get information about available clustering algorithms | |
| """ | |
| return { | |
| "algorithms": { | |
| "kmeans": { | |
| "name": "K-Means", | |
| "description": "Partition-based clustering that minimizes within-cluster variance", | |
| "parameters": ["n_clusters"], | |
| "pros": ["Fast", "Scalable", "Works well with spherical clusters"], | |
| "cons": ["Requires specifying k", "Sensitive to outliers", "Assumes spherical clusters"] | |
| }, | |
| "dbscan": { | |
| "name": "DBSCAN", | |
| "description": "Density-based clustering that can find arbitrarily shaped clusters", | |
| "parameters": ["eps", "min_samples"], | |
| "pros": ["No need to specify k", "Finds arbitrary shapes", "Identifies outliers"], | |
| "cons": ["Sensitive to parameters", "Struggles with varying densities"] | |
| }, | |
| "gmm": { | |
| "name": "Gaussian Mixture Model", | |
| "description": "Probabilistic model that assumes data is generated from mixture of Gaussians", | |
| "parameters": ["n_clusters"], | |
| "pros": ["Soft clustering", "Provides probabilities", "Flexible cluster shapes"], | |
| "cons": ["Computationally expensive", "Sensitive to initialization"] | |
| }, | |
| "spectral": { | |
| "name": "Spectral Clustering", | |
| "description": "Uses eigenvalues of similarity matrix to reduce dimensions before clustering", | |
| "parameters": ["n_clusters"], | |
| "pros": ["Works with non-convex clusters", "Graph-based approach"], | |
| "cons": ["Computationally expensive", "Memory intensive for large datasets"] | |
| } | |
| }, | |
| "metrics": { | |
| "silhouette_score": "Measures how similar a point is to its own cluster vs other clusters (-1 to 1, higher is better)", | |
| "davies_bouldin_score": "Average similarity ratio of each cluster with most similar cluster (lower is better)", | |
| "calinski_harabasz_score": "Ratio of between-cluster to within-cluster dispersion (higher is better)" | |
| } | |
| } | |
| async def download_clustering_model(user_id: str): | |
| """ | |
| Download the trained clustering model as PKL file. | |
| If no PKL exists, creates one from the JSON model data. | |
| """ | |
| from fastapi.responses import FileResponse, Response | |
| try: | |
| paths = get_user_paths(user_id) | |
| models_dir = paths.get("models", paths["base"] / "models") | |
| models_dir.mkdir(parents=True, exist_ok=True) | |
| # Try to find existing PKL | |
| pkl_files = list(models_dir.glob("clustering_model_*.pkl")) | |
| if pkl_files: | |
| latest_pkl = max(pkl_files, key=lambda p: p.stat().st_mtime) | |
| return FileResponse( | |
| path=str(latest_pkl), | |
| media_type='application/octet-stream', | |
| filename=latest_pkl.name | |
| ) | |
| # No PKL found — create one from the active JSON model | |
| active_json = models_dir / "active_clustering.json" | |
| if not active_json.exists(): | |
| # Try any clustering JSON | |
| json_files = list(models_dir.glob("clustering_*.json")) | |
| if json_files: | |
| active_json = max(json_files, key=lambda p: p.stat().st_mtime) | |
| else: | |
| raise HTTPException(status_code=404, detail="No clustering model found. Train a model first.") | |
| with open(active_json, 'r') as f: | |
| model_json = json.load(f) | |
| # Reconstruct a PKL with scaler from JSON data | |
| scaler_mean = model_json.get('scaler_mean') | |
| scaler_scale = model_json.get('scaler_scale') | |
| reconstructed_scaler = None | |
| if scaler_mean and scaler_scale: | |
| reconstructed_scaler = StandardScaler() | |
| reconstructed_scaler.mean_ = np.array(scaler_mean) | |
| reconstructed_scaler.scale_ = np.array(scaler_scale) | |
| reconstructed_scaler.var_ = np.array(scaler_scale) ** 2 | |
| reconstructed_scaler.n_features_in_ = len(scaler_mean) | |
| pkl_data = { | |
| 'algorithm': model_json.get('algorithm', 'kmeans'), | |
| 'n_clusters': model_json.get('n_clusters', 3), | |
| 'scaler': reconstructed_scaler, | |
| 'feature_columns': model_json.get('feature_columns', []), | |
| 'centroids_scaled': np.array(model_json.get('centroids_scaled', [])) if model_json.get('centroids_scaled') else None, | |
| 'labels': np.array(model_json.get('labels', [])), | |
| 'model_id': model_json.get('model_id', 'unknown'), | |
| 'created_at': model_json.get('created_at', ''), | |
| 'silhouette_score': model_json.get('silhouette_score', 0), | |
| } | |
| # Save as PKL for future use | |
| model_id = model_json.get('model_id', f"clustering_{uuid.uuid4().hex[:8]}") | |
| pkl_filename = f"clustering_model_{model_id}.pkl" | |
| pkl_path = models_dir / pkl_filename | |
| with open(pkl_path, 'wb') as f: | |
| pickle.dump(pkl_data, f) | |
| logger.info(f"✅ Created PKL from JSON: {pkl_path}") | |
| return FileResponse( | |
| path=str(pkl_path), | |
| media_type='application/octet-stream', | |
| filename=pkl_filename | |
| ) | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| logger.error(f"❌ Download model error: {e}") | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| async def download_clustered_data(user_id: str): | |
| """ | |
| Download the cleaned data with cluster assignments. | |
| If no clustered CSV exists, creates one from the original data + model labels. | |
| """ | |
| from fastapi.responses import FileResponse, Response | |
| try: | |
| paths = get_user_paths(user_id) | |
| files_dir = paths.get("files", paths["base"] / "files") | |
| models_dir = paths.get("models", paths["base"] / "models") | |
| files_dir.mkdir(parents=True, exist_ok=True) | |
| models_dir.mkdir(parents=True, exist_ok=True) | |
| # Try to find existing clustered data | |
| csv_files = list(files_dir.glob("clustered_data_*.csv")) | |
| if csv_files: | |
| latest_csv = max(csv_files, key=lambda p: p.stat().st_mtime) | |
| return FileResponse( | |
| path=str(latest_csv), | |
| media_type='text/csv', | |
| filename=latest_csv.name | |
| ) | |
| # No clustered data found — try to reconstruct from model labels + original data | |
| active_json = models_dir / "active_clustering.json" | |
| if not active_json.exists(): | |
| json_files = list(models_dir.glob("clustering_*.json")) | |
| if json_files: | |
| active_json = max(json_files, key=lambda p: p.stat().st_mtime) | |
| else: | |
| raise HTTPException(status_code=404, detail="No clustered data found. Run clustering first.") | |
| with open(active_json, 'r') as f: | |
| model_json = json.load(f) | |
| labels = model_json.get('labels', []) | |
| feature_columns = model_json.get('feature_columns', []) | |
| if not labels: | |
| raise HTTPException(status_code=404, detail="No cluster labels found. Run clustering first.") | |
| # Try to find the most recent uploaded CSV to attach labels to | |
| all_csv = list(files_dir.glob("*.csv")) | |
| # Exclude already-clustered files | |
| original_csvs = [f for f in all_csv if not f.name.startswith("clustered_data_") and not f.name.startswith("cleaned_")] | |
| if original_csvs: | |
| # Use the most recent original CSV | |
| data_file = max(original_csvs, key=lambda p: p.stat().st_mtime) | |
| df = pd.read_csv(data_file) | |
| if len(df) == len(labels): | |
| df['Cluster'] = labels | |
| df['Cluster_Name'] = [f'Cluster_{l}' if l >= 0 else 'Noise' for l in labels] | |
| # Create and save clustered CSV | |
| model_id = model_json.get('model_id', f"clustering_{uuid.uuid4().hex[:8]}") | |
| clustered_filename = f"clustered_data_{model_id}.csv" | |
| clustered_path = files_dir / clustered_filename | |
| df.to_csv(clustered_path, index=False) | |
| logger.info(f"✅ Reconstructed clustered data: {clustered_path}") | |
| return FileResponse( | |
| path=str(clustered_path), | |
| media_type='text/csv', | |
| filename=clustered_filename | |
| ) | |
| # Last resort: just return labels as CSV | |
| labels_df = pd.DataFrame({'sample_index': range(len(labels)), 'cluster': labels}) | |
| csv_content = labels_df.to_csv(index=False) | |
| return Response( | |
| content=csv_content, | |
| media_type='text/csv', | |
| headers={"Content-Disposition": "attachment; filename=cluster_labels.csv"} | |
| ) | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| logger.error(f"❌ Download data error: {e}") | |
| raise HTTPException(status_code=500, detail=str(e)) | |
| async def download_clustering_code(user_id: str): | |
| """ | |
| Download a complete clustering project as a ZIP file. | |
| Includes: model.pkl, clustered_data.csv, predict_cluster.py, train_clustering.py, | |
| visualize_clusters.py, api_server.py, charts/, README.md, Dockerfile, requirements.txt | |
| """ | |
| from fastapi.responses import StreamingResponse | |
| try: | |
| paths = get_user_paths(user_id) | |
| models_dir = paths.get("models", paths["base"] / "models") | |
| files_dir = paths.get("files", paths["base"] / "files") | |
| models_dir.mkdir(parents=True, exist_ok=True) | |
| files_dir.mkdir(parents=True, exist_ok=True) | |
| # Find latest clustering model PKL | |
| pkl_files = list(models_dir.glob("clustering_model_*.pkl")) | |
| latest_pkl = max(pkl_files, key=lambda p: p.stat().st_mtime) if pkl_files else None | |
| # If no PKL, try to create from JSON | |
| if not latest_pkl: | |
| active_json = models_dir / "active_clustering.json" | |
| if not active_json.exists(): | |
| json_files = list(models_dir.glob("clustering_*.json")) | |
| if json_files: | |
| active_json = max(json_files, key=lambda p: p.stat().st_mtime) | |
| else: | |
| raise HTTPException(status_code=404, detail="No clustering model found. Train a model first.") | |
| with open(active_json, 'r') as f: | |
| model_json = json.load(f) | |
| # Reconstruct PKL from JSON | |
| scaler_mean = model_json.get('scaler_mean') | |
| scaler_scale = model_json.get('scaler_scale') | |
| reconstructed_scaler = None | |
| if scaler_mean and scaler_scale: | |
| reconstructed_scaler = StandardScaler() | |
| reconstructed_scaler.mean_ = np.array(scaler_mean) | |
| reconstructed_scaler.scale_ = np.array(scaler_scale) | |
| reconstructed_scaler.var_ = np.array(scaler_scale) ** 2 | |
| reconstructed_scaler.n_features_in_ = len(scaler_mean) | |
| pkl_data_dict = { | |
| 'algorithm': model_json.get('algorithm', 'kmeans'), | |
| 'n_clusters': model_json.get('n_clusters', 3), | |
| 'scaler': reconstructed_scaler, | |
| 'feature_columns': model_json.get('feature_columns', []), | |
| 'centroids_scaled': np.array(model_json.get('centroids_scaled', [])) if model_json.get('centroids_scaled') else None, | |
| 'labels': np.array(model_json.get('labels', [])), | |
| 'model_id': model_json.get('model_id', 'unknown'), | |
| 'created_at': model_json.get('created_at', ''), | |
| 'silhouette_score': model_json.get('silhouette_score', 0), | |
| } | |
| model_id = model_json.get('model_id', f"clustering_{uuid.uuid4().hex[:8]}") | |
| pkl_filename = f"clustering_model_{model_id}.pkl" | |
| latest_pkl = models_dir / pkl_filename | |
| with open(latest_pkl, 'wb') as f: | |
| pickle.dump(pkl_data_dict, f) | |
| logger.info(f"✅ Created PKL from JSON for ZIP: {latest_pkl}") | |
| # Load model metadata from PKL | |
| clustering_meta = {} | |
| pkl_data = {} | |
| try: | |
| with open(latest_pkl, 'rb') as f: | |
| pkl_data = pickle.load(f) | |
| clustering_meta = { | |
| 'algorithm': pkl_data.get('algorithm', 'kmeans'), | |
| 'n_clusters': pkl_data.get('n_clusters', 3), | |
| 'silhouette_score': pkl_data.get('silhouette_score', 0), | |
| 'feature_columns': pkl_data.get('feature_columns', []), | |
| 'cluster_profiles': pkl_data.get('cluster_profiles', {}), | |
| } | |
| except Exception as e: | |
| logger.warning(f"Could not load PKL metadata: {e}") | |
| # Find latest clustered data CSV | |
| csv_files = list(files_dir.glob("clustered_data_*.csv")) | |
| cleaned_data_path = max(csv_files, key=lambda p: p.stat().st_mtime) if csv_files else None | |
| # If no clustered CSV, try to reconstruct | |
| if not cleaned_data_path: | |
| try: | |
| labels = pkl_data.get('labels', []) if pkl_data else [] | |
| if hasattr(labels, 'tolist'): | |
| labels = labels.tolist() | |
| all_csv = list(files_dir.glob("*.csv")) | |
| original_csvs = [f for f in all_csv if not f.name.startswith("clustered_data_") and not f.name.startswith("cleaned_")] | |
| if original_csvs and labels: | |
| data_file = max(original_csvs, key=lambda p: p.stat().st_mtime) | |
| df = pd.read_csv(data_file) | |
| if len(df) == len(labels): | |
| df['Cluster'] = labels | |
| df['Cluster_Name'] = [f'Cluster_{l}' if l >= 0 else 'Noise' for l in labels] | |
| clustered_filename = f"clustered_data_reconstructed.csv" | |
| cleaned_data_path = files_dir / clustered_filename | |
| df.to_csv(cleaned_data_path, index=False) | |
| logger.info(f"✅ Reconstructed clustered data for ZIP: {cleaned_data_path}") | |
| except Exception as e: | |
| logger.warning(f"Could not reconstruct clustered data: {e}") | |
| if cleaned_data_path: | |
| try: | |
| import pandas as _pd | |
| _df = _pd.read_csv(cleaned_data_path) | |
| clustering_meta['n_samples'] = len(_df) | |
| except Exception: | |
| pass | |
| # Load charts if available | |
| charts_data = None | |
| try: | |
| # Check for active clustering charts stored as JSON | |
| charts_json = models_dir / "active_clustering_charts.json" | |
| if charts_json.exists(): | |
| with open(charts_json, 'r') as f: | |
| charts_data = json.load(f) | |
| else: | |
| # Try model_persistence charts | |
| try: | |
| from ml.model_persistence import model_persistence | |
| charts_data = model_persistence.get_charts(user_id) | |
| except Exception: | |
| pass | |
| except Exception as e: | |
| logger.warning(f"Could not load charts: {e}") | |
| # Generate ZIP | |
| from ml.ml_code_generator import generate_clustering_code_zip | |
| zip_buffer = generate_clustering_code_zip( | |
| pkl_path=latest_pkl, | |
| cleaned_data_path=cleaned_data_path, | |
| charts_data=charts_data, | |
| clustering_meta=clustering_meta, | |
| ) | |
| return StreamingResponse( | |
| zip_buffer, | |
| media_type="application/zip", | |
| headers={"Content-Disposition": "attachment; filename=clustering_project.zip"} | |
| ) | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| logger.error(f"❌ Download clustering code error: {e}") | |
| raise HTTPException(status_code=500, detail=str(e)) |