File size: 3,711 Bytes
a4a265d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
from typing import Dict, List, Any
import numpy as np
from sklearn.pipeline import Pipeline
from sklearn.compose import ColumnTransformer
from sklearn.impute import SimpleImputer
from sklearn.preprocessing import OneHotEncoder, RobustScaler
from sklearn.feature_selection import SelectKBest, f_classif
from category_encoders import TargetEncoder
from src.monitoring.logger import get_logger

logger = get_logger(__name__)

class FeatureEngineeringPipeline:
    """Creates an adaptive sklearn ColumnTransformer based on feature cardinality."""
    
    @staticmethod
    def build_preprocessor(feature_layout: Dict[str, List[str]], is_tree_model: bool) -> ColumnTransformer:
        """
        Builds the preprocessing ColumnTransformer.
        Applies RobustScaler conditionally.
        Uses TargetEncoder for high cardinality features.
        
        Runtime Anomaly Fallbacks Enabled:
        - Out-of-Vocabulary Categories -> mapped to unseen ('ignore' / 'value')
        - Null-Heavy inference payloads -> mediated safely by SimpleImputer bounds.
        """
        transformers = []
        
        # 1. Numerical Pipeline
        if feature_layout.get("numerical"):
            from sklearn.preprocessing import PolynomialFeatures
            num_steps = [('imputer', SimpleImputer(strategy='median'))]
            
            # Feature Synthesis: Automated interaction discovery
            # (degree=2, interaction_only=True helps find non-linear relationships between features)
            # We limit this to datasets where it won't explode feature count too much
            if len(feature_layout["numerical"]) >= 2 and len(feature_layout["numerical"]) <= 15:
                logger.info("Enabling Feature Synthesis (Interactions) for numerical columns.")
                num_steps.append(('synth', PolynomialFeatures(degree=2, interaction_only=True, include_bias=False)))
            
            if not is_tree_model:
                num_steps.append(('scaler', RobustScaler()))
            transformers.append(('num', Pipeline(num_steps), feature_layout["numerical"]))
            
        # 2. Low Cardinality Categorical Pipeline (OneHotEncoding)
        if feature_layout.get("low_cardinality"):
            low_card_steps = [
                ('imputer', SimpleImputer(strategy='constant', fill_value='Missing')),
                ('ohe', OneHotEncoder(handle_unknown='ignore', sparse_output=False))
            ]
            transformers.append(('low_cat', Pipeline(low_card_steps), feature_layout["low_cardinality"]))
            
        # 3. High Cardinality Categorical Pipeline (TargetEncoding)
        if feature_layout.get("high_cardinality"):
            high_card_steps = [
                ('imputer', SimpleImputer(strategy='constant', fill_value='Missing')),
                ('target_enc', TargetEncoder(handle_unknown='value'))
            ]
            transformers.append(('high_cat', Pipeline(high_card_steps), feature_layout["high_cardinality"]))
            
        preprocessor = ColumnTransformer(transformers=transformers, n_jobs=-1, remainder='drop')
        logger.info(f"Preprocessor built with {len(transformers)} transformers. (Tree Model: {is_tree_model})")
        return preprocessor
        
    @staticmethod
    def get_adaptive_feature_limit(n_samples: int, n_features: int) -> int:
        """Adaptive feature limit to strictly control memory."""
        computed_limit = int(np.sqrt(n_samples * n_features))
        limit = min(50, computed_limit)
        limit = max(1, limit) # Ensure at least 1 feature is selected
        logger.info(f"Adaptive feature limit set to {limit} limit (samples={n_samples}, features={n_features})")
        return limit