File size: 4,081 Bytes
c2b9b65
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
import pandas as pd
import numpy as np
import plotly.express as px
from pycaret.regression import *
from utilities import is_numeric


class Multivariate_Models:
    def __init__(self, data, forecasting_horizon, target_column, date_column, time_interval, fold=3,
                 metric_to_opt="RMSLE", session_id=42):
        
        #Pre-Processing Steps
        data[date_column] = pd.to_datetime(data[date_column])
        if time_interval == "daily":
            data['month'] = data[date_column].map(lambda x: str(x.month))
            data['day'] = data[date_column].map(lambda x: str(x.day))
        elif time_interval == "weekly":
            data['week'] = data[date_column].map(lambda x: str(x.week))
            data['month'] = data[date_column].map(lambda x: str(x.month))
        elif time_interval == "monthly":
            data['month'] = data[date_column].map(lambda x: str(x.month))
        elif time_interval == "yearly":
            pass


        data['trend'] = np.arange(1, data.shape[0]+1)
        data.drop(columns=[date_column], inplace=True)

        all_cols = set(data.columns) - set((date_column, target_column))
        num_features = [col for col in all_cols if is_numeric(data[col])]
        cat_features = [col for col in all_cols if col not in num_features]
        data[cat_features] = data[cat_features].astype('category', copy=True)

        self.train_data = data.iloc[:-forecasting_horizon]
        self.test_data = data.iloc[-forecasting_horizon:]

        # Initialization
        self.s = setup(data=self.train_data, test_data=self.test_data, target=target_column,
                       numeric_features=num_features, categorical_features=cat_features,
                       categorical_imputation="mode", numeric_imputation="knn",
                       fold_strategy="timeseries", fold=fold,
                       session_id=session_id)
        
        self.forcasting_horizon = forecasting_horizon
        self.metric_to_opt=metric_to_opt
        self.target_column = target_column
        self.full_data = data.copy()
        self.num_features = num_features
        self.cat_features = cat_features

    def get_stats(self):
        return self.cat_features, self.num_features, self.train_data.describe()
    
    def train_models(self):
        best_model = compare_models(sort=self.metric_to_opt)
        model_comp_df = pull()
        self.best_model = tune_model(best_model, search_algorithm='grid')
        best_mod_param_df = pd.DataFrame.from_dict(self.best_model.get_params(), orient='index')
        return model_comp_df, best_mod_param_df, self.best_model

    def get_plots(self):
        plots_dict = {}
        plots_name_dict = {'residuals': 'Residuals Plot', 'error': 'Prediction Error Plot',
                           'cooks': 'Cooks Distance Plot', 'rfe': 'Recursive Feat. Selection',
                           'learning': 'Learning Curve', 'vc': 'Validation Curve',
                           'manifold': 'Manifold Learning', 'feature_all': 'Feature Importance',
                           'tree': 'Decision Tree'} #'parameter': 'Model Hyperparameter', 
        for k,v in plots_name_dict.items():
            # All plots are not available for all models
            try:
                plots_dict[v] = plot_model(self.best_model, plot = k, save=True)
            except Exception as e:
                print(f"[PLOT EXCEPTION] ==> {type(e).__name__} : {e}")
        return plots_dict
    

    
    def get_predictions(self):
        self.y_hat = predict_model(self.best_model)
        return self.y_hat[["prediction_label"]]

    def plot_insample_and_forecast(self):
        self.y_hat_all = predict_model(self.best_model, data=self.full_data)

        outsam = px.line(self.y_hat, y=[self.target_column, 'prediction_label'], x='trend',
                      title="Actual vs Out-of-Sample Forecast")
        
        insam = px.line(self.y_hat_all, y=[self.target_column, 'prediction_label'], x='trend',
                      title="Actual vs Forecast (Insample)")
        
        return insam, outsam