Spaces:
Configuration error
Configuration error
| """Shared, leakage-safe feature engineering (FEAT-01..04). | |
| This module is the single source of truth for turning a raw PaySim-shaped | |
| transaction (or batch of transactions) into the model's feature vector. It | |
| is imported by both `training/train.py` (batch, DataFrame-based) and the | |
| future Phase 3 scoring service (single-transaction), so feature logic never | |
| drifts between train and serve. | |
| Design notes: | |
| - Raw balance columns (`oldbalanceOrg`, `newbalanceOrig`, `oldbalanceDest`, | |
| `newbalanceDest`) are never fed to the model directly -- only ratios and | |
| consistency flags derived from them (FEAT-03), because fraudulent | |
| TRANSFER/CASH_OUT rows trivially zero `newbalanceOrig`, which would let a | |
| model "cheat" on a feature that is actually just restating the label. | |
| - `isFlaggedFraud` is never referenced here -- EDA showed it has ~0.19% | |
| recall against `isFraud` and is a simulation artifact, not a fraud | |
| signal (see reports/EDA_REPORT.md section 5). | |
| - Per-account velocity features are computed as *prior-only* aggregates | |
| (expanding count/sum strictly before the current row for that account), | |
| so a transaction never sees its own future -- this is what keeps | |
| `engineer_features` safe to use directly ahead of a step-based | |
| train/test split (MODEL-03). | |
| """ | |
| from __future__ import annotations | |
| import pandas as pd | |
| EPSILON = 1.0 # avoids divide-by-zero without materially distorting ratios | |
| TRANSACTION_TYPES = ["CASH_IN", "CASH_OUT", "DEBIT", "PAYMENT", "TRANSFER"] | |
| # Final feature columns handed to every model. Keeping this list explicit | |
| # (rather than "everything numeric in the frame") means adding a new | |
| # intermediate column to `engineer_features` never silently changes the | |
| # model's input shape. | |
| FEATURE_COLUMNS = [ | |
| "amount", | |
| "amount_to_oldbalanceOrg_ratio", | |
| "amount_to_oldbalanceDest_ratio", | |
| "orig_balance_delta_ratio", | |
| "dest_balance_delta_ratio", | |
| "orig_balance_consistent", | |
| "dest_balance_consistent", | |
| "orig_zero_after_flag", | |
| "dest_zero_stays_zero_flag", | |
| "dest_is_merchant", | |
| "orig_equals_dest", | |
| "orig_prior_txn_count", | |
| "orig_prior_txn_amount_mean", | |
| "dest_prior_txn_count", | |
| ] + [f"type_{t}" for t in TRANSACTION_TYPES] | |
| # PaySim generates fraud rows such that amount == oldbalanceOrg (and thus | |
| # newbalanceOrig == 0) in ~98% of fraud cases, but *never* exactly for | |
| # legitimate TRANSFER/CASH_OUT rows -- a simulation artifact, not a | |
| # real-world fraud behavior. `amount_to_oldbalanceOrg_ratio` (~1.0 for | |
| # fraud) and `orig_balance_delta_ratio` (mathematically the same relationship | |
| # restated) let a tree model memorize this exact-equality quirk rather than | |
| # learn a generalizable pattern, which is what drove PR-AUC to a suspicious | |
| # 1.0000 in the full feature set. FEATURE_COLUMNS_CONSERVATIVE excludes both | |
| # so the comparison can show how much of the full set's performance is real | |
| # signal versus this artifact. | |
| LEAKAGE_PRONE_COLUMNS = ["amount_to_oldbalanceOrg_ratio", "orig_balance_delta_ratio"] | |
| FEATURE_COLUMNS_CONSERVATIVE = [ | |
| c for c in FEATURE_COLUMNS if c not in LEAKAGE_PRONE_COLUMNS | |
| ] | |
| def _encode_type(df: pd.DataFrame) -> pd.DataFrame: | |
| """FEAT-02: one-hot encode transaction type against a fixed vocabulary. | |
| A fixed vocabulary (rather than `pd.get_dummies` alone) guarantees the | |
| same columns exist at scoring time even if a single transaction can | |
| only ever be one type. | |
| """ | |
| for t in TRANSACTION_TYPES: | |
| df[f"type_{t}"] = (df["type"] == t).astype("int8") | |
| return df | |
| def _balance_ratio_features(df: pd.DataFrame) -> pd.DataFrame: | |
| """FEAT-01 / FEAT-03: ratios and consistency flags, never raw balances.""" | |
| old_orig = df["oldbalanceOrg"] | |
| new_orig = df["newbalanceOrig"] | |
| old_dest = df["oldbalanceDest"] | |
| new_dest = df["newbalanceDest"] | |
| amount = df["amount"] | |
| df["amount_to_oldbalanceOrg_ratio"] = amount / (old_orig + EPSILON) | |
| df["amount_to_oldbalanceDest_ratio"] = amount / (old_dest + EPSILON) | |
| df["orig_balance_delta_ratio"] = (old_orig - new_orig) / (old_orig + EPSILON) | |
| df["dest_balance_delta_ratio"] = (new_dest - old_dest) / (old_dest + EPSILON) | |
| df["orig_balance_consistent"] = ( | |
| (old_orig - amount - new_orig).abs() < 0.01 | |
| ).astype("int8") | |
| df["dest_balance_consistent"] = ( | |
| (old_dest + amount - new_dest).abs() < 0.01 | |
| ).astype("int8") | |
| df["orig_zero_after_flag"] = ((old_orig > 0) & (new_orig == 0)).astype("int8") | |
| df["dest_zero_stays_zero_flag"] = ( | |
| (old_dest == 0) & (new_dest == 0) & (amount > 0) | |
| ).astype("int8") | |
| return df | |
| def _origin_dest_mismatch_features(df: pd.DataFrame) -> pd.DataFrame: | |
| """FEAT-01: origin/destination mismatch patterns.""" | |
| df["dest_is_merchant"] = df["nameDest"].str.startswith("M").astype("int8") | |
| df["orig_equals_dest"] = (df["nameOrig"] == df["nameDest"]).astype("int8") | |
| return df | |
| def _velocity_features(df: pd.DataFrame) -> pd.DataFrame: | |
| """FEAT-01: per-account transaction velocity, computed leakage-safe. | |
| `df` must already be sorted ascending by `step` (the caller's | |
| responsibility -- `engineer_features` enforces it). Each account's | |
| velocity aggregates use `groupby(...).cumcount()` / `.cumsum().shift()`, | |
| which by construction only ever reflect that account's *earlier* rows | |
| in step order, never the current or future ones. | |
| """ | |
| orig_group = df.groupby("nameOrig")["amount"] | |
| df["orig_prior_txn_count"] = orig_group.cumcount() | |
| prior_amount_sum = orig_group.cumsum() - df["amount"] | |
| df["orig_prior_txn_amount_mean"] = ( | |
| prior_amount_sum / df["orig_prior_txn_count"].replace(0, pd.NA) | |
| ).fillna(0.0) | |
| df["dest_prior_txn_count"] = df.groupby("nameDest").cumcount() | |
| return df | |
| def engineer_features(df: pd.DataFrame) -> pd.DataFrame: | |
| """Add all model features to a copy of `df`. | |
| `df` must contain the raw PaySim columns: step, type, amount, nameOrig, | |
| oldbalanceOrg, newbalanceOrig, nameDest, oldbalanceDest, newbalanceDest. | |
| Returns a new DataFrame sorted by `step` (stable) with every column in | |
| `FEATURE_COLUMNS` added, plus all original columns preserved. | |
| Velocity features require the *full* history for an account to be | |
| present in `df` at once (e.g. the whole training set, or -- at scoring | |
| time -- a query result of that account's own transaction history from | |
| `transactions`) since they aggregate strictly prior same-account rows. | |
| """ | |
| df = df.sort_values("step", kind="mergesort").reset_index(drop=True) | |
| df = _encode_type(df) | |
| df = _balance_ratio_features(df) | |
| df = _origin_dest_mismatch_features(df) | |
| df = _velocity_features(df) | |
| return df | |