import os import pickle import pandas as pd import numpy as np from sklearn.model_selection import train_test_split from sklearn.preprocessing import LabelEncoder, StandardScaler from sklearn.ensemble import RandomForestClassifier from sklearn.metrics import classification_report, accuracy_score, roc_auc_score, confusion_matrix import xgboost as xgb def train_delay_model(): cleaned_csv_path = os.path.join("Data", "cleaned_data.csv") models_dir = "Models" os.makedirs(models_dir, exist_ok=True) if not os.path.exists(cleaned_csv_path): print(f"Error: Cleaned data not found at {cleaned_csv_path}. Run data_cleaning.py first.") return print("Loading cleaned dataset...") df = pd.read_csv(cleaned_csv_path) # Define Target Variable target_col = "late_delivery_risk" if target_col not in df.columns: print(f"Error: Target column '{target_col}' not found in dataset.") return print(f"Target distribution:\n{df[target_col].value_counts(normalize=True)}") # 1. PREVENT DATA LEAKAGE: Identify and drop post-shipment features leakage_cols = [ "days_for_shipping_real", "delivery_status", "shipping_date", "order_status" ] # Drop columns that are unique identifiers or text fields that can't be easily modeled id_text_cols = [ "customer_fname", "customer_lname", "customer_street", "customer_zipcode", "order_zipcode", "product_name", "category_name", "department_name" ] drop_cols = leakage_cols + id_text_cols drop_cols = [col for col in drop_cols if col in df.columns] print(f"Dropping post-shipment (leakage) and text/id columns: {drop_cols}") df_features = df.drop(columns=drop_cols) # 2. FEATURE ENGINEERING print("Performing advanced feature engineering...") # Convert order_date to datetime to extract temporal features df_features["order_date"] = pd.to_datetime(df_features["order_date"]) df_features["order_year"] = df_features["order_date"].dt.year df_features["order_month"] = df_features["order_date"].dt.month df_features["order_day"] = df_features["order_date"].dt.day df_features["order_hour"] = df_features["order_date"].dt.hour df_features["order_dayofweek"] = df_features["order_date"].dt.dayofweek df_features["is_weekend"] = df_features["order_dayofweek"].isin([5, 6]).astype(int) # Cross-feature interactions # 1. Check if shipping is domestic or international (cross-border) df_features["is_domestic"] = (df_features["customer_country"] == df_features["order_country"]).astype(int) # 2. Check if shipping is within same state df_features["is_same_state"] = (df_features["customer_state"] == df_features["order_state"]).astype(int) # 3. Check if shipping is within same city df_features["is_same_city"] = (df_features["customer_city"] == df_features["order_city"]).astype(int) # 4. Calculate discount amount in currency df_features["discount_amount"] = df_features["sales"] * df_features["order_item_discount_rate"] # 5. Calculate unit price per item df_features["price_per_item"] = df_features["sales"] / (df_features["order_item_quantity"] + 1e-5) # Drop original order_date df_features = df_features.drop(columns=["order_date"]) # 3. ENCODE CATEGORICAL VARIABLES print("Encoding categorical features...") categorical_cols = [ "type", "customer_segment", "customer_city", "customer_state", "customer_country", "market", "order_city", "order_state", "order_country", "order_region", "shipping_mode" ] label_encoders = {} for col in categorical_cols: if col in df_features.columns: df_features[col] = df_features[col].fillna("Unknown").astype(str) le = LabelEncoder() df_features[col] = le.fit_transform(df_features[col]) label_encoders[col] = le # Save encoders for future inference encoders_path = os.path.join(models_dir, "label_encoders.pkl") with open(encoders_path, "wb") as f: pickle.dump(label_encoders, f) print("Saved label encoders.") # 4. TRAIN-TEST SPLIT X = df_features.drop(columns=[target_col]) y = df_features[target_col] print(f"Number of features: {X.shape[1]}") X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y) print(f"Train shape: {X_train.shape}, Test shape: {X_test.shape}") # 5. SCALE NUMERICAL FEATURES # Identify numerical columns to scale numerical_cols = [ "benefit_per_order", "sales_per_customer", "latitude", "longitude", "order_item_discount", "order_item_discount_rate", "order_item_product_price", "order_item_profit_ratio", "order_item_quantity", "sales", "order_item_total", "order_profit_per_order", "product_price", "days_for_shipping_scheduled", "discount_amount", "price_per_item" ] numerical_cols = [col for col in numerical_cols if col in X.columns] scaler = StandardScaler() X_train[numerical_cols] = scaler.fit_transform(X_train[numerical_cols]) X_test[numerical_cols] = scaler.transform(X_test[numerical_cols]) # Save scaler for future inference scaler_path = os.path.join(models_dir, "scaler.pkl") with open(scaler_path, "wb") as f: pickle.dump(scaler, f) print("Saved feature scaler.") # 6. MODEL TRAINING # Model A: Random Forest (Baseline) print("\nTraining Random Forest Classifier (Baseline)...") rf_model = RandomForestClassifier(n_estimators=100, max_depth=12, random_state=42, n_jobs=-1) rf_model.fit(X_train, y_train) rf_preds = rf_model.predict(X_test) print("\nRandom Forest Test Evaluation:") print(f"Accuracy: {accuracy_score(y_test, rf_preds):.4f}") print(f"ROC-AUC: {roc_auc_score(y_test, rf_model.predict_proba(X_test)[:, 1]):.4f}") print("\nClassification Report:") print(classification_report(y_test, rf_preds)) # Model B: XGBoost Classifier (Advanced - Tuned) print("\nTraining Tuned XGBoost Classifier...") xgb_model = xgb.XGBClassifier( n_estimators=300, max_depth=8, learning_rate=0.05, subsample=0.8, colsample_bytree=0.8, random_state=42, eval_metric="logloss", n_jobs=-1 ) xgb_model.fit(X_train, y_train) xgb_preds = xgb_model.predict(X_test) xgb_probs = xgb_model.predict_proba(X_test)[:, 1] print("\nTuned XGBoost Test Evaluation:") print(f"Accuracy: {accuracy_score(y_test, xgb_preds):.4f}") print(f"ROC-AUC: {roc_auc_score(y_test, xgb_probs):.4f}") print("\nClassification Report:") print(classification_report(y_test, xgb_preds)) print("\nConfusion Matrix:") print(confusion_matrix(y_test, xgb_preds)) # Feature Importance (XGBoost) importances = xgb_model.feature_importances_ feat_imp = pd.Series(importances, index=X.columns).sort_values(ascending=False) print("\nTop 10 Feature Importances (XGBoost):") print(feat_imp.head(10)) # 7. SAVE THE BEST MODEL xgb_model_path = os.path.join(models_dir, "best_xgb_model.json") xgb_model.save_model(xgb_model_path) print(f"\nSaved best XGBoost model to {xgb_model_path}") # Save a pickle version of the model for easy loading xgb_pkl_path = os.path.join(models_dir, "best_xgb_model.pkl") with open(xgb_pkl_path, "wb") as f: pickle.dump(xgb_model, f) print(f"Saved best XGBoost model as pickle to {xgb_pkl_path}") print("Model training completed successfully!") if __name__ == "__main__": train_delay_model()