File size: 24,355 Bytes
ee7d7b9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
"""
ML Reports - Real AutoML Charts & Visualizations
Uses ONLY real data from trained models
Includes task-specific charts: Regression, Classification, NLP
"""

def generate_predictive_report_v2(user_id: str, df, profiler) -> dict:
    """
    ๐Ÿ”ฎ PREDICTIVE REPORT - Real ML Model Visualizations
    
    For REGRESSION:
    - Actual vs Predicted scatter plot
    - Residual distribution
    - Rยฒ and error metrics
    
    For CLASSIFICATION:
    - Confusion matrix data
    - Class distribution
    - Precision/Recall/F1 metrics
    
    For NLP:
    - Feature importance from text
    - Sentiment/category distribution
    """
    from datetime import datetime
    import numpy as np
    
    CHART_COLORS = ['#14B8A6', '#22C55E', '#3B82F6', '#F59E0B', '#8B5CF6', '#EC4899', '#EF4444']
    
    sections = []
    n = len(df)
    
    # Load AutoML model
    automl_info = None
    automl_engine = None
    
    try:
        from ml.model_persistence import model_persistence
        from ml.automl_engine import automl_engine as engine
        
        metadata = model_persistence.get_metadata(user_id)
        if metadata:
            automl_info = {
                'model_name': metadata.model_name,
                'task_type': metadata.task_type,
                'target_column': metadata.target_column,
                'metrics': metadata.metrics or {},
                'version': metadata.version
            }
            # Load the model and check if it has a trained model
            if engine.load(user_id) and engine.model is not None:
                automl_engine = engine
    except Exception as e:
        print(f"AutoML load error: {e}")
    
    # NO MODEL - Show instructions
    if not automl_info:
        sections.append({
            "title": "โš ๏ธ No ML Model Trained",
            "content": f"""Train an AutoML model to see real predictions and ML charts.

Steps:
1. Go to Data Hub
2. Upload your CSV/Excel file
3. Click "๐Ÿค– Auto ML Train"
4. Select target column
5. Wait for training

Your Data: {n:,} records, {len(profiler.numeric_cols)} numeric, {len(profiler.categorical_cols)} categorical""",
            "data": []
        })
        return {
            "title": "๐Ÿ”ฎ Predictive Report - Train Model First",
            "generatedAt": datetime.now().isoformat(),
            "dataSource": "uploaded_files",
            "sections": sections,
            "reportType": "predictive"
        }
    
    # MODEL EXISTS - Robust Data Extraction
    model_name = automl_info.get('model_name', 'Unknown Model')
    task_type = automl_info.get('task_type', 'unknown')
    
    # Robust target column extraction
    target_col = automl_info.get('target_column')
    if not target_col and automl_engine:
        target_col = getattr(automl_engine, 'target_column', 'Unknown Target')
    if not target_col:
        target_col = "Unknown Target"
        
    metrics = automl_info.get('metrics') or {}
    # Fallback to engine metrics if metadata empty
    if not metrics and automl_engine:
        metrics = getattr(automl_engine, 'metrics', {})
    
    # ---------------------------------------------
    # PRE-LOAD REAL CHARTS (To decide on legacy sections)
    # ---------------------------------------------
    real_charts = {}
    if automl_engine:
        try:
            from ml.model_persistence import model_persistence
            # 1. Try to load saved charts
            saved_charts = model_persistence.get_charts(user_id)
            if saved_charts:
                real_charts = saved_charts
        except:
            pass
            
    # ---------------------------------------------
    # GENERATE DYNAMIC INSIGHTS (Natural Language)
    # ---------------------------------------------
    primary_metric = "Accuracy" if task_type == "classification" else "Rยฒ Score"
    primary_score = metrics.get('accuracy', metrics.get('r2_score', 0))
    
    # Format score for display
    if primary_metric == "Rยฒ Score" and primary_score < -1:
        score_display = "(Negative Rยฒ)"
    else:
        score_display = f"{primary_score:.1%}"
    
    performance_text = "moderate"
    if primary_score > 0.85: performance_text = "excellent"
    elif primary_score > 0.7: performance_text = "good" 
    elif primary_score < 0.5: performance_text = "poor"
    
    # Get top features for narrative
    top_features_text = ""
    if automl_engine and hasattr(automl_engine, '_get_importance'):
        fi_list = automl_engine._get_importance(automl_engine.model)
        if fi_list:
            top_3 = [f.get('feature', '').replace('_', ' ').title() for f in fi_list[:3]]
            if top_3:
                top_features_text = f"The most influential factors driving these predictions are {', '.join(top_3)}."

    # Section 0: Executive Summary
    sections.append({
        "title": "๐Ÿ“ Executive Summary",
        "content": f"""The {model_name} model has been trained for {task_type.upper()} tasks on the target '{target_col}'.

Performance Assessment:
The model demonstrates {performance_text} performance with a {primary_metric} of {score_display}. {top_features_text}

Recommendation:
{("Reliable for automated decision making." if primary_score > 0.8 else "Use for guidance, but verify critical cases manually.")}""",
        "data": {"model": model_name, "score": primary_score, "quality": performance_text}
    })
    
    # Section 1: Model Overview
    sections.append({
        "title": f"๐Ÿค– Model Configuration",
        "content": f"""โ€ข Algorithm: {model_name}
โ€ข Task Type: {task_type.title()}
โ€ข Target Variable: {target_col}
โ€ข Training Records: {n:,}
โ€ข Model Version: v{automl_info.get('version', '1')}""",
        "data": {"model": model_name, "task": task_type, "target": target_col}
    })
    
    # Section 2: Performance Metrics Chart (Legacy - ONLY if no real charts, or if explicit metric charts missing)
    # We hide this if we have the new visual charts to avoid duplication
    if metrics and not real_charts:
        metrics_chart = []
        metric_names = {
            'accuracy': 'Accuracy', 'precision': 'Precision', 'recall': 'Recall', 'f1': 'F1 Score', 'f1_score': 'F1 Score',
            'r2_score': 'Rยฒ Score', 'mae': 'MAE (Error)', 'rmse': 'RMSE (Error)'
        }
        
        for k, v in metrics.items():
            if isinstance(v, (int, float)) and k in metric_names:
                display_val = v
                # Normalize 0-1 metrics to percentages for display, keep errors as is
                if k not in ['mae', 'rmse'] and 0 <= v <= 1:
                     display_val = v * 100
                
                metrics_chart.append({
                    "name": metric_names.get(k, k),
                    "value": round(display_val, 2),
                    "color": CHART_COLORS[len(metrics_chart) % len(CHART_COLORS)]
                })
        
        if metrics_chart:
            sections.append({
                "title": "๐Ÿ“Š Performance Metrics",
                "content": f"Key performance indicators for {model_name}. Higher is better (except error metrics).",
                "data": metrics_chart,
                "chartType": "horizontal_bar"
            })
    
    # Section 3: Feature Importance (Legacy - ONLY if no real charts)
    feature_importance = None
    if automl_engine and not real_charts:
        # Try to get feature importance from the model
        if hasattr(automl_engine, '_get_importance'):
            fi_list = automl_engine._get_importance(automl_engine.model)
            if fi_list:
                # Convert list of dicts to chart data
                fi_chart = []
                for i, item in enumerate(fi_list[:10]):
                    feat = item.get('feature', f'Feature {i}')
                    imp = item.get('importance', 0)
                    fi_chart.append({
                        "name": feat.replace('_', ' ').title()[:15],
                        "value": round(imp * 100, 2),
                        "color": CHART_COLORS[i % len(CHART_COLORS)]
                    })
                if fi_chart:
                    sections.append({
                        "title": "๐ŸŽฏ Key Drivers (Feature Importance)",
                        "content": "These features have the strongest impact on the target variable. Focus on optimizing these factors to influence outcomes.",
                        "data": fi_chart,
                        "chartType": "horizontal_bar"
                    })
    
    # =============================================
    # TASK-SPECIFIC REAL ML CHARTS (Images)
    # =============================================
    
    if automl_engine:
        try:
            # 2. Fallback: Regenerate if missing (Logic for legacy or prediction-only)
            if not real_charts and target_col in df.columns:
                # Use a larger sample for chart generation to ensure representative visuals
                chart_sample_size = min(500, len(df))
                sample = df.head(chart_sample_size).copy()
                preds = automl_engine.predict(sample)
                
                # Get probabilities if available (for ROC/PR curves)
                probs = None
                if task_type == 'classification' and hasattr(automl_engine.model, 'predict_proba'):
                    try:
                        probs = automl_engine.model.predict_proba(sample)
                    except:
                        pass
                
                y_true = sample[target_col].values
                from ml.chart_generator import generate_ml_charts
                
                # Generate comprehensive chart suite on the fly
                real_charts = generate_ml_charts(
                    task_type=task_type,
                    y_test=y_true,
                    y_pred=preds,
                    y_proba=probs,
                    model_name=model_name,
                    class_names=automl_engine.classes_ if hasattr(automl_engine, 'classes_') else None
                )
            
            # 3. Add charts to sections
            # Map charts to friendly titles and descriptions
            chart_descriptions = {
                'confusion_matrix': ('Confusion Matrix', 'Visualizes how often the model confuses different classes. Diagonal values represent correct predictions.'),
                'roc_curve': ('ROC Curve', 'Shows the trade-off between True Positive Rate and False Positive Rate. AUC score closer to 1.0 is better.'),
                'feature_importance': ('Feature Importance', 'Ranks features by their influence on the model\'s decisions.'),
                'actual_vs_predicted': ('Actual vs Predicted', 'Comparison of model predictions against real values. Points along the diagonal line indicate perfect accuracy.'),
                'residuals_analysis': ('Residuals Analysis', 'Analyzes prediction errors to check for bias or patterns.'),
                'class_distribution': ('Class Distribution', 'Compare predicted class frequencies against actual frequencies.'),
                'precision_recall': ('Precision-Recall Curve', 'Trade-off between Precision and Recall, crucial for imbalanced datasets.'),
                'prediction_overview': ('Prediction Overview', 'Visualizes predictions against actual values across the dataset.'),
                'error_distribution': ('Error Distribution', 'Histogram of prediction errors. Narrower distribution centered at 0 means better accuracy.'),
                'distribution_grid': ('Feature Distributions', 'Histograms showing the spread of data for top numeric features.'),
                'boxplot_grid': ('Feature Box Plots', 'Box plots showing outliers and quartiles for numeric features.'),
                'correlation_heatmap': ('Correlation Heatmap', 'Heatmap showing how features correlate with each other.'),
                'model_comparison': ('Model Comparison', 'Performance comparison of all trained models.')
            }
            
            for chart_key, base64_img in real_charts.items():
                if chart_key in chart_descriptions:
                    title, desc = chart_descriptions[chart_key]
                    sections.append({
                        "title": f"๐Ÿ“Š {title}",
                        "content": desc,
                        "data": {"image": base64_img},
                        "chartType": "image"
                    })
            
            # 4. Handle Prediction-Only Mode (No Target Column AND No Saved Charts)
            if not real_charts and target_col not in df.columns:
                # No ground truth - Prediction Only Mode
                if preds is not None:
                    # Add Prediction Distribution (Pie Chart)
                    if task_type == 'classification':
                        from collections import Counter
                        class_counts = Counter([str(p) for p in preds])
                        class_chart = [{"name": str(k)[:15], "value": v, "color": CHART_COLORS[i % len(CHART_COLORS)]} 
                                      for i, (k, v) in enumerate(class_counts.most_common(10))]
                        sections.append({
                            "title": "๐Ÿ“Š Predicted Class Distribution",
                            "content": f"Distribution of predicted classes for the {len(preds)} analyzed records.",
                            "data": class_chart,
                            "chartType": "pie"
                        })
                    
                    # Add Sample Predictions
                    pred_chart = []
                    for i, p in enumerate(preds[:10]):
                        val = p if isinstance(p, (int, float, str)) else str(p)
                        pred_chart.append({
                            "name": f"Rec {i+1}",
                            "value": float(p) if isinstance(p, (int, float)) else i,
                            "label": str(val)[:15],
                            "color": CHART_COLORS[i % len(CHART_COLORS)]
                        })
                    sections.append({
                        "title": "๐Ÿ”Ž Prediction Samples",
                        "content": "A glimpse of the model's output on your data.",
                        "data": pred_chart,
                        "chartType": "bar"
                    })

        except Exception as e:
            print(f"Real ML Chart Generation Error: {e}")
            import traceback
            traceback.print_exc()
    
    # Section: Strategic Recommendations (Action Item for Frontend)
    sections.append({
        "title": "โšก Strategic Recommendations",
        "content": f"""1. Verify the model's predictions on new data using the 'Predict' tab.
2. Focus on the key drivers ({top_3[0] if 'top_3' in locals() and top_3 else 'identified features'}) to influence outcomes.
3. {("Since the model is highly accurate, consider automating workflows." if primary_score > 0.8 else "Use these predictions as a support tool for human subject matter experts.")}""",
        "data": []
    })

    return {
        "title": f"๐Ÿ”ฎ AI Predictive Report - {model_name}",
        "generatedAt": datetime.now().isoformat(),
        "dataSource": "AutoML Engine",
        "sections": sections,
        "reportType": "predictive"
    }


def generate_anomaly_report_v2(user_id: str, df, profiler) -> dict:
    """
    โš ๏ธ ANOMALY REPORT - Real Statistical Analysis
    
    Charts:
    - Outlier distribution (box plot data)
    - Missing values bar chart
    - Data quality gauge
    - Distribution analysis
    """
    from datetime import datetime
    import numpy as np
    
    CHART_COLORS = ['#14B8A6', '#22C55E', '#3B82F6', '#F59E0B', '#8B5CF6', '#EC4899', '#EF4444']
    
    sections = []
    n = len(df)
    
    # Check for model context
    model_info = None
    try:
        from ml.model_persistence import model_persistence
        metadata = model_persistence.get_metadata(user_id)
        if metadata:
            model_info = {
                'model_name': metadata.model_name,
                'target_column': metadata.target_column,
                'task_type': metadata.task_type
            }
    except:
        pass
    
    # Section 1: Overview
    overview = f"""Anomaly Detection Analysis
โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”
Records: {n:,}
Numeric: {len(profiler.numeric_cols)}
Categorical: {len(profiler.categorical_cols)}"""
    
    if model_info:
        overview += f"\n\nML Model: {model_info['model_name']} ({model_info['task_type']})"
    
    sections.append({
        "title": "โš ๏ธ Anomaly Detection",
        "content": overview,
        "data": {"records": n}
    })
    
    # Section 2: Distribution Analysis with Box Plot Data
    box_plot_data = []
    total_outliers = 0
    outlier_details = []
    
    for i, col in enumerate(profiler.numeric_cols[:6]):
        try:
            vals = df[col].dropna()
            if len(vals) < 5:
                continue
            
            q1 = float(np.percentile(vals, 25))
            q3 = float(np.percentile(vals, 75))
            iqr = q3 - q1
            lower = q1 - 1.5 * iqr
            upper = q3 + 1.5 * iqr
            
            outliers = vals[(vals < lower) | (vals > upper)]
            outlier_count = len(outliers)
            total_outliers += outlier_count
            
            col_name = col.replace('_', ' ').title()[:12]
            
            box_plot_data.append({
                "name": col_name,
                "min": round(float(vals.min()), 2),
                "q1": round(q1, 2),
                "median": round(float(vals.median()), 2),
                "q3": round(q3, 2),
                "max": round(float(vals.max()), 2),
                "outliers": outlier_count,
                "color": CHART_COLORS[i % len(CHART_COLORS)]
            })
            
            if outlier_count > 0:
                pct = (outlier_count / len(vals)) * 100
                outlier_details.append({
                    "name": col_name,
                    "value": outlier_count,
                    "percentage": round(pct, 1),
                    "color": "#EF4444" if pct > 5 else "#F59E0B"
                })
        except:
            continue
    
    if box_plot_data:
        sections.append({
            "title": "๐Ÿ“Š Distribution Analysis",
            "content": "Statistical distribution (Q1, Median, Q3):",
            "data": box_plot_data,
            "chartType": "box"
        })
    
    # Section 3: Outlier Count Chart
    if outlier_details:
        sections.append({
            "title": "๐Ÿ” Outliers Found",
            "content": f"Total: {total_outliers} outliers detected using IQR method",
            "data": outlier_details,
            "chartType": "bar"
        })
    else:
        sections.append({
            "title": "โœ… No Outliers",
            "content": "No significant outliers detected.",
            "data": []
        })
    
    # Section 4: Missing Values
    missing_data = []
    total_missing = 0
    for col in df.columns[:12]:
        missing = int(df[col].isna().sum())
        total_missing += missing
        if missing > 0:
            pct = (missing / n) * 100
            missing_data.append({
                "name": str(col)[:10],
                "value": missing,
                "percentage": round(pct, 1),
                "color": "#EF4444" if pct > 10 else "#F59E0B" if pct > 2 else "#22C55E"
            })
    
    if missing_data:
        sections.append({
            "title": "๐Ÿ“Š Missing Values",
            "content": f"Total: {total_missing:,} missing values",
            "data": missing_data[:8],
            "chartType": "horizontal_bar"
        })
    
    # Section 5: Duplicates
    dup_count = int(df.duplicated().sum())
    if dup_count > 0:
        sections.append({
            "title": "๐Ÿ”„ Duplicates",
            "content": f"Found {dup_count:,} duplicate rows ({dup_count/n*100:.1f}%)",
            "data": [
                {"name": "Unique", "value": n - dup_count, "color": "#22C55E"},
                {"name": "Duplicates", "value": dup_count, "color": "#EF4444"}
            ],
            "chartType": "pie"
        })
    
    # Section 6: Data Quality Score
    issues = total_outliers + dup_count + total_missing
    max_issues = n * len(df.columns) if n > 0 else 1
    quality_score = max(0, min(100, int(100 - (issues / max_issues * 100))))
    
    status = "โœ… Excellent" if quality_score >= 80 else "๐ŸŸก Good" if quality_score >= 60 else "๐ŸŸ  Needs Work" if quality_score >= 40 else "๐Ÿ”ด Poor"
    
    sections.append({
        "title": "๐Ÿ“‹ Data Quality Score",
        "content": f"""Score: {quality_score}/100 - {status}

Issues Found:
โ€ข Outliers: {total_outliers}
โ€ข Missing: {total_missing}
โ€ข Duplicates: {dup_count}""",
        "data": [{"name": "Quality", "value": quality_score, "max": 100, "color": "#22C55E" if quality_score >= 70 else "#F59E0B"}],
        "chartType": "gauge"
    })
    
    # Section 7: Recommended Actions
    actions = []
    
    # ===========================================
    # SECTION 8: ML Anomaly Visualization (Isolation Forest)
    # ===========================================
    if len(profiler.numeric_cols) >= 2 and n >= 10:
        try:
            from sklearn.ensemble import IsolationForest
            from ml.chart_generator import generate_ml_charts
            
            # Prepare data
            X = df[profiler.numeric_cols].dropna().values
            # Run Isolation Forest
            iso = IsolationForest(contamination=0.05, random_state=42)
            y_pred = iso.fit_predict(X) # 1 for normal, -1 for anomaly
            
            # Convert to cluster labels (0=Anomaly, 1=Normal)
            # IsolationForest returns -1 for anomaly, 1 for normal
            # Let's map -1 -> 0 (Anomaly), 1 -> 1 (Normal) for better visualization colors
            cluster_labels = np.where(y_pred == -1, 0, 1)
            
            charts = generate_ml_charts(
                task_type='clustering',
                y_test=cluster_labels, # Dummy ground truth (same as pred) to satisfy signature
                y_pred=cluster_labels,
                X_test=X,
                feature_names=profiler.numeric_cols,
                class_names=['Anomaly', 'Normal']
            )
            
            if 'cluster_scatter' in charts:
                sections.append({
                    "title": "๐Ÿ“Š Anomaly Visualization (PCA)",
                    "content": "2D projection of data using Principal Component Analysis (PCA). Points in Cluster 0 (Anomaly) are statistically distinct from normal patterns.",
                    "data": {"image": charts['cluster_scatter']},
                    "chartType": "image"
                })
                
                # Check for other useful charts like 'cluster_box_plots'
                if 'cluster_box_plots' in charts:
                    sections.append({
                        "title": "๐Ÿ“ฆ Feature Distribution by Anomaly Status",
                        "content": "Comparison of feature distributions between Normal (1) and Anomalous (0) records.",
                        "data": {"image": charts['cluster_box_plots']},
                        "chartType": "image"
                    })
                    
        except Exception as e:
            print(f"Anomaly ML visual error: {e}")

    if total_outliers > 0:
        actions.append(f"1. Investigate the {total_outliers} detected outliers in the 'Outliers Found' section.")
    if total_missing > 0:
        actions.append(f"2. Consider imputing or removing the {total_missing} missing values.")
    if dup_count > 0:
        actions.append(f"3. Remove {dup_count} duplicate records to prevent data leakage.")
    
    if not actions:
        actions.append("1. Data quality is excellent. Proceed with analysis or modeling.")
        
    sections.append({
        "title": "โšก Recommended Actions",
        "content": "\n".join(actions),
        "data": []
    })
    
    return {
        "title": "โš ๏ธ Anomaly Detection Report",
        "generatedAt": datetime.now().isoformat(),
        "dataSource": "uploaded_files",
        "sections": sections,
        "reportType": "anomaly"
    }