Spaces:
Sleeping
Sleeping
Download src/utils/feature_engineering.py from useifabdelhady/FraudDetection: direct link, hf CLI and curl.
- Browser
- Download file 6.16 kB
-
https://huggingface.co/spaces/useifabdelhady/FraudDetection/resolve/main/src/utils/feature_engineering.py
- Command line
-
hf download hf://spaces/useifabdelhady/FraudDetection/src/utils/feature_engineering.py
-
curl -L -o feature_engineering.py https://huggingface.co/spaces/useifabdelhady/FraudDetection/resolve/main/src/utils/feature_engineering.py
6.16 kB
| import pandas as pd | |
| import numpy as np | |
| import logging | |
| from sklearn.preprocessing import StandardScaler | |
| from src.config.parameters import FEATURE_PARAMS | |
| logger = logging.getLogger(__name__) | |
| class FeatureEngineer: | |
| """Feature engineering class for fraud detection""" | |
| def __init__(self): | |
| self.scaler = StandardScaler() | |
| self.legit_amount_mean = None | |
| self.amount_bin_edges = None | |
| self.top_corr_features = None | |
| def engineer_features(self, df: pd.DataFrame, is_training: bool = True) -> pd.DataFrame: | |
| """Apply all feature engineering steps""" | |
| logger.info("Starting feature engineering") | |
| df_processed = df.copy() | |
| df_processed = self._scale_features(df_processed, is_training) | |
| df_processed = self._create_temporal_features(df_processed) | |
| df_processed = self._create_amount_features(df_processed, is_training) | |
| df_processed = self._create_v_aggregated_features(df_processed) | |
| df_processed = self._create_feature_interactions(df_processed, is_training) # This is now fixed | |
| df_processed = self._cleanup_columns(df_processed) | |
| # The reorder step is no longer needed as the DataPreprocessor handles final column alignment | |
| logger.info(f"Feature engineering completed. New shape: {df_processed.shape}") | |
| return df_processed | |
| def _scale_features(self, df: pd.DataFrame, is_training: bool) -> pd.DataFrame: | |
| """Scale Amount and Time features with one scaler""" | |
| if is_training: | |
| df[['scaled_amount', 'scaled_time']] = self.scaler.fit_transform(df[['Amount', 'Time']]) | |
| else: | |
| if not hasattr(self.scaler, 'scale_'): | |
| raise RuntimeError("Scaler has not been fitted. Please run the training pipeline first.") | |
| df[['scaled_amount', 'scaled_time']] = self.scaler.transform(df[['Amount', 'Time']]) | |
| return df | |
| def _create_temporal_features(self, df: pd.DataFrame) -> pd.DataFrame: | |
| """Create temporal features from Time""" | |
| df['hour_of_day'] = (df['Time'] % 86400) // 3600 | |
| df['time_bin'] = pd.cut( | |
| df['hour_of_day'], | |
| bins=FEATURE_PARAMS['time_bins'], | |
| labels=FEATURE_PARAMS['time_labels'], | |
| include_lowest=True | |
| ) | |
| df = pd.get_dummies(df, columns=['time_bin'], drop_first=True) | |
| return df | |
| def _create_amount_features(self, df: pd.DataFrame, is_training: bool) -> pd.DataFrame: | |
| """Create amount-based features""" | |
| if is_training: | |
| if 'Class' in df.columns: | |
| self.legit_amount_mean = df[df['Class'] == 0]['scaled_amount'].mean() | |
| else: | |
| self.legit_amount_mean = df['scaled_amount'].mean() | |
| # FIX: More robust way to create and save bin edges for perfect consistency. | |
| _, self.amount_bin_edges = pd.qcut( | |
| df['scaled_amount'], | |
| q=FEATURE_PARAMS['amount_quantiles'], | |
| labels=FEATURE_PARAMS['amount_labels'], | |
| retbins=True, # Return the bin edges | |
| duplicates='drop' | |
| ) | |
| # This check is important for prediction mode | |
| if self.legit_amount_mean is None or self.amount_bin_edges is None: | |
| raise RuntimeError("Amount features artifacts (mean, bins) are not available. Run training first.") | |
| df['amount_deviation'] = df['scaled_amount'] - self.legit_amount_mean | |
| df['amount_bin'] = pd.cut( | |
| df['scaled_amount'], | |
| bins=self.amount_bin_edges, | |
| labels=FEATURE_PARAMS['amount_labels'], | |
| include_lowest=True | |
| ) | |
| df = pd.get_dummies(df, columns=['amount_bin'], drop_first=True) | |
| return df | |
| def _create_v_aggregated_features(self, df: pd.DataFrame) -> pd.DataFrame: | |
| """Create aggregated features from V1-V28""" | |
| v_columns = FEATURE_PARAMS['v_columns'] | |
| df['mean_V'] = df[v_columns].mean(axis=1) | |
| df['std_V'] = df[v_columns].std(axis=1) | |
| return df | |
| def _create_feature_interactions(self, df: pd.DataFrame, is_training: bool) -> pd.DataFrame: | |
| """Create feature interactions based on correlation with target""" | |
| # FIX: Step 1 - Identify top features ONLY during training. | |
| if is_training and 'Class' in df.columns: | |
| numerical_cols = ['scaled_time', 'scaled_amount', 'hour_of_day', | |
| 'amount_deviation', 'mean_V', 'std_V'] + FEATURE_PARAMS['v_columns'] | |
| # Ensure all numerical columns exist before calculating correlation | |
| existing_numerical_cols = [col for col in numerical_cols if col in df.columns] | |
| corr = df[existing_numerical_cols + ['Class']].corr()['Class'].abs().sort_values(ascending=False) | |
| self.top_corr_features = corr[1:FEATURE_PARAMS['top_corr_features_count']+1].index.tolist() | |
| # This check is important for prediction mode | |
| if self.top_corr_features is None: | |
| raise RuntimeError("Top correlated features for interactions are not set. Run training first.") | |
| # FIX: Step 2 - Create the interaction features in BOTH training and prediction modes. | |
| # This block is no longer inside the `if is_training:` condition. | |
| for i, f1 in enumerate(self.top_corr_features): | |
| for f2 in self.top_corr_features[i+1:]: | |
| # Ensure source columns exist before creating the interaction term | |
| if f1 in df.columns and f2 in df.columns: | |
| col_name = f'{f1}_{f2}_interaction' | |
| df[col_name] = df[f1] * df[f2] | |
| return df | |
| def _cleanup_columns(self, df: pd.DataFrame) -> pd.DataFrame: | |
| """Remove original Time and Amount columns""" | |
| columns_to_drop = ['Time', 'Amount', 'hour_of_day'] # hour_of_day is intermediate | |
| existing_columns_to_drop = [col for col in columns_to_drop if col in df.columns] | |
| if existing_columns_to_drop: | |
| df.drop(existing_columns_to_drop, axis=1, inplace=True) | |
| return df |