{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AMERICAN EXPRESS - DEFAULT PREDICTION 2022\n# Ultra Fast Adversarial Validation Combined with Feature SHAP Importances\n\n**Developer: Paulo A. Buoro**\n\n**Goal:**\n\nThe goal of this notebook is to provide a fast way to do adversarial validation and, at the same time, try to avoid the drop of features which although unstable accordingly to the AV may be worth keep due to their power.\n\nIts ultra speed is due to the high Cross-Validation AUC delta as the AV model does not need to be extremely accurate.\n\n**Main Hyperparameters:**\n\nOne should treat mainly the AUC and SHAP value thresholds as hyperparameters and they should therefore be tuned to reach desirable final results.\n\n**Output csv files:**\n\nADV_VAL_drop.csv: Contains the list of features to drop accordingly to the AV.\n\nADV_VAL_high_shap_maybe_keep.csv: Contains the list of features in ADV_VAL_drop.csv| which should be worth keeping due to their high SHAP Importance.","metadata":{}},{"cell_type":"code","source":"# SETUP\n\nimport numpy as np\nimport pandas as pd\nimport datetime\nimport xgboost as xgb\nfrom sklearn.metrics import roc_auc_score\nimport shap\nfrom sklearn.model_selection import train_test_split\nimport gc\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# LOAD DATA\n\ntest_data = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet')\ntest_data = test_data.drop_duplicates(subset='customer_ID', keep='last')\ntest_data = test_data.drop(columns=['S_2', 'customer_ID'])\n\ntrain_data = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\ntrain_data = train_data.drop_duplicates(subset='customer_ID', keep='last')\ntrain_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\ntrain_data = train_data.merge(train_labels, on='customer_ID', how='inner')\ntrain_labels = train_data['target'].to_numpy().ravel()\ntrain_data = train_data.drop(columns=['S_2', 'customer_ID', 'target'])\n\ntrain_data['adv_val_target'] = 1\ntest_data['adv_val_target'] = 0\n\ndata = pd.concat([train_data, test_data], axis=0)\ndel train_data, test_data\n\nx_train = data.drop(columns='adv_val_target').reset_index(drop=True)\ny_train = data[['adv_val_target']]\n\ndel data\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T21:45:51.757569Z","iopub.execute_input":"2022-08-19T21:45:51.758008Z","iopub.status.idle":"2022-08-19T21:47:33.614504Z","shell.execute_reply.started":"2022-08-19T21:45:51.757919Z","shell.execute_reply":"2022-08-19T21:47:33.613404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****ADVERSARIAL VALIDATION CLASS****","metadata":{}},{"cell_type":"code","source":"class adversarial_validation():\n    '''\n    ADVERSARIAL VALIDATION CLASS\n    '''\n    def __init__(self, x_train: pd.DataFrame, y_train: pd.DataFrame, train_labels: np.array, \n                 test_size: int, n_features_drop: int, auc_thrd: int, shap_thrd: int) -> None:\n        '''\n        Adversarial Validation Class\n        x_train (dataframe): Train dataframe.\n        y_train (dataframe): Adversarial Validation target dataframe.\n        train_labels (array): Competition target.\n        test_size (int): % of test size [0, 1].\n        n_features_drop (int): # features to drop per iteration.\n        auc_thrd (int): AUC threshold to stop dropping features.\n        shap_thrd (int): SHAP Importance threshold to keep even if unstable.\n        '''    \n        self.x_train = x_train\n        self.y_train = y_train\n        self.train_labels = train_labels\n        self.test_size = test_size\n        self.n_features_drop = n_features_drop\n        self.auc_thrd = auc_thrd\n        self.shap_thrd = shap_thrd\n\n    def adv_val_loop(self) -> None:\n        '''\n        Adversarial Validation Loop\n        '''\n        model = xgb.XGBClassifier(tree_method='gpu_hist', grow_policy='lossguide', random_state=56,\n                                    min_child_weight=50, subsample=0.65, colsample_bytree=0.6, colsample_bylevel=0.6, colsample_bynode=0.55,\n                                    learning_rate=0.015, max_depth=8, max_bin=320, max_leaves=0)\n\n        drop_features_ite = []\n        all_feature_importances = []\n        features_to_use_ite = list(self.x_train.columns)\n        roc_auc = 1\n        \n        while roc_auc > self.auc_thrd:\n            print('--------------------------------')\n            x_train_cv, x_val_cv, y_train_cv, y_val_cv = train_test_split(self.x_train[features_to_use_ite], self.y_train, test_size=self.test_size, \n                                                                          shuffle=True, random_state=42)\n            y_train_cv = y_train_cv.to_numpy().ravel()\n            y_val_cv = y_val_cv.to_numpy().ravel()\n            \n            print(f'iteration x_train cv shape: {x_train_cv.shape}')\n            model.set_params(**{'n_estimators': 20000})\n            es = xgb.callback.EarlyStopping(rounds=100, min_delta=0.01, maximize=True, save_best=True)\n            model.fit(x_train_cv, y_train_cv, eval_set=[(x_val_cv, y_val_cv)], eval_metric='auc', callbacks=[es], verbose=500)\n\n            y_pred = model.predict_proba(x_val_cv)[:,1]\n            roc_auc = roc_auc_score(y_val_cv, y_pred)\n            print(f'ROC AUC: {roc_auc}')\n            \n            feature_importances = pd.DataFrame()\n            feature_importances['importance'] = model.feature_importances_\n            feature_importances['features'] = features_to_use_ite\n            feature_importances = feature_importances.sort_values(by='importance', ascending=False)\n            all_feature_importances.append(feature_importances)\n            drop_features_ite += feature_importances['features'].to_list()[:self.n_features_drop]\n            features_to_use_ite = [feat for feat in features_to_use_ite if feat not in drop_features_ite]\n            print(f'Features dropped: {drop_features_ite}')\n        pd.DataFrame(drop_features_ite, columns=['Feature']).to_csv(f'./ADV_VAL_drop.csv')\n        \n        print('----------------------------------------------------------')\n        print(f'Estimator Classifier for Shap FI / Start time: {datetime.datetime.now().strftime(\"%Y-%m-%d %H:%M:%S\")}')\n        model.set_params(**{'n_estimators': 800})\n        train_labels_index = self.y_train[self.y_train.adv_val_target == 1].index\n        print('Train shape:', self.x_train.iloc[train_labels_index].shape)\n        model.fit(self.x_train.iloc[train_labels_index], self.train_labels)\n        print(f'Estimator Classifier for Shap FI / End time: {datetime.datetime.now().strftime(\"%Y-%m-%d %H:%M:%S\")}')\n    \n        # SHAP Values\n        print('----------------------------------------------------------')\n        print(f'SHAP Calculation / Start time: {datetime.datetime.now().strftime(\"%Y-%m-%d %H:%M:%S\")}')\n        features_name = self.x_train.columns\n        shap_values = shap.TreeExplainer(model).shap_values(self.x_train.iloc[train_labels_index])\n        features_shap_abs = abs(shap_values)\n        features_shap_mean_abs = features_shap_abs.mean(axis=0)\n        features_shap_mean_abs = pd.DataFrame(features_shap_mean_abs).transpose()\n        features_shap_mean_abs.columns = features_name\n        features_shap_mean_abs = features_shap_mean_abs.transpose().reset_index()\n        features_shap_mean_abs.columns = ['Feature', 'shap']\n        features_shap_mean_abs = features_shap_mean_abs.sort_values(by=[\"shap\"], ascending=False).reset_index(drop=True)\n        features_shap_mean_abs.to_csv('./features_shap_mean_abs.csv')\n        print(f'SHAP Calculation / End time: {datetime.datetime.now().strftime(\"%Y-%m-%d %H:%M:%S\")}')\n        \n        features_high_shap = features_shap_mean_abs[features_shap_mean_abs.shap > self.shap_thrd]['Feature'].tolist()\n        features_to_maybe_keep = [feat for feat in drop_features_ite if feat in features_high_shap]\n        pd.DataFrame(features_to_maybe_keep, columns=['Feature']).to_csv(f'./ADV_VAL_high_shap_maybe_keep.csv')\n        print(f'Features high shap maybe keep: {features_to_maybe_keep}')\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-19T21:47:33.617032Z","iopub.execute_input":"2022-08-19T21:47:33.617459Z","iopub.status.idle":"2022-08-19T21:47:33.637515Z","shell.execute_reply.started":"2022-08-19T21:47:33.617410Z","shell.execute_reply":"2022-08-19T21:47:33.636439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ADV_VAL = adversarial_validation(x_train, y_train, train_labels, test_size=0.4, n_features_drop=10, auc_thrd=0.6, shap_thrd=0.02)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T21:47:33.639059Z","iopub.execute_input":"2022-08-19T21:47:33.639429Z","iopub.status.idle":"2022-08-19T21:47:33.649625Z","shell.execute_reply.started":"2022-08-19T21:47:33.639388Z","shell.execute_reply":"2022-08-19T21:47:33.648718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ADV_VAL.adv_val_loop()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T21:47:33.652290Z","iopub.execute_input":"2022-08-19T21:47:33.652642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_to_drop = pd.read_csv('./ADV_VAL_drop.csv', index_col=0)\nfeatures_to_keep = pd.read_csv('./ADV_VAL_high_shap_maybe_keep.csv', index_col=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}