{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport lightgbm as lgb\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import train_test_split\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n\n%matplotlib inline\nplt.style.use('bmh')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T09:58:40.430783Z","iopub.execute_input":"2022-07-09T09:58:40.431440Z","iopub.status.idle":"2022-07-09T09:58:42.160935Z","shell.execute_reply.started":"2022-07-09T09:58:40.431386Z","shell.execute_reply":"2022-07-09T09:58:42.160188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Please refer to this  notebook I created for getting the feature rankings: \n**https://www.kaggle.com/code/ishaan45/amex-feature-engineering-ranking**\n\n**Any suggestions are most welcome!**","metadata":{}},{"cell_type":"markdown","source":"### Selecting feaures","metadata":{}},{"cell_type":"code","source":"feats_rank = pd.read_csv('../input/amexfeatureranking/all_feat_ranking.csv')\nfeats_rank['mean_shap_value'] = feats_rank['mean_shap_value'].round(4)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:12.676511Z","iopub.execute_input":"2022-07-03T19:31:12.676894Z","iopub.status.idle":"2022-07-03T19:31:12.70181Z","shell.execute_reply.started":"2022-07-03T19:31:12.67686Z","shell.execute_reply":"2022-07-03T19:31:12.700767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LGB Based Feature importance scores distribution\nsns.histplot(feats_rank['tree_split'], kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:12.704917Z","iopub.execute_input":"2022-07-03T19:31:12.705337Z","iopub.status.idle":"2022-07-03T19:31:13.077832Z","shell.execute_reply.started":"2022-07-03T19:31:12.705307Z","shell.execute_reply":"2022-07-03T19:31:13.076893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shap values distribution for all features\nsns.histplot(feats_rank['mean_shap_value'], bins=50)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:13.080021Z","iopub.execute_input":"2022-07-03T19:31:13.080618Z","iopub.status.idle":"2022-07-03T19:31:13.366164Z","shell.execute_reply.started":"2022-07-03T19:31:13.080573Z","shell.execute_reply":"2022-07-03T19:31:13.365313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Getting relevant features from raw data","metadata":{}},{"cell_type":"code","source":"def get_relevant_features(imp_df,top_n=0,stratergy='shap',val_thresh=0):\n    if top_n > 0:\n        top_feats = imp_df.sort_values(by='shap_rank' if stratergy =='shap' else 'lgb_rank', ascending=True).head(top_n)\n    else:\n        top_feats = imp_df[imp_df['tree_split' if stratergy =='lgb' else 'mean_shap_value'] > val_thresh]\n    \n    top_feat_names = top_feats['feature_name'].values.tolist()\n    return top_feat_names","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:13.367304Z","iopub.execute_input":"2022-07-03T19:31:13.367734Z","iopub.status.idle":"2022-07-03T19:31:13.374602Z","shell.execute_reply.started":"2022-07-03T19:31:13.367705Z","shell.execute_reply":"2022-07-03T19:31:13.373714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_list = get_relevant_features(feats_rank, top_n=204, stratergy='shap')\nraw_feats = [feat for feat in feats_list if not ('_mean' in feat or '_std' in feat)]","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:13.375598Z","iopub.execute_input":"2022-07-03T19:31:13.376312Z","iopub.status.idle":"2022-07-03T19:31:13.38901Z","shell.execute_reply.started":"2022-07-03T19:31:13.376244Z","shell.execute_reply":"2022-07-03T19:31:13.387691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_feats = [feat for feat in feats_list if '_mean' in feat]\nstd_feats = [feat for feat in feats_list if '_std' in feat]","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:13.39256Z","iopub.execute_input":"2022-07-03T19:31:13.393237Z","iopub.status.idle":"2022-07-03T19:31:13.400287Z","shell.execute_reply.started":"2022-07-03T19:31:13.393191Z","shell.execute_reply":"2022-07-03T19:31:13.398974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_cols = list(map(lambda x:x[:-5], mean_feats))\nstd_cols = list(map(lambda x:x[:-4], std_feats))\ncols_to_load = list(set(raw_feats + mean_cols + std_cols))","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:13.58274Z","iopub.execute_input":"2022-07-03T19:31:13.583175Z","iopub.status.idle":"2022-07-03T19:31:13.58938Z","shell.execute_reply.started":"2022-07-03T19:31:13.583139Z","shell.execute_reply":"2022-07-03T19:31:13.588603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total number of columns to be loaded: {len(cols_to_load)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:13.900448Z","iopub.execute_input":"2022-07-03T19:31:13.90085Z","iopub.status.idle":"2022-07-03T19:31:13.907293Z","shell.execute_reply.started":"2022-07-03T19:31:13.900819Z","shell.execute_reply":"2022-07-03T19:31:13.90608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading specific columns","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_feather('../input/amexfeather/train_data.ftr', columns=cols_to_load + ['customer_ID','target'])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T08:34:25.507856Z","iopub.execute_input":"2022-07-09T08:34:25.508263Z","iopub.status.idle":"2022-07-09T08:34:25.624791Z","shell.execute_reply.started":"2022-07-09T08:34:25.508228Z","shell.execute_reply":"2022-07-09T08:34:25.623830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Initial shape of train data: {train_df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:37.897412Z","iopub.execute_input":"2022-07-03T19:31:37.898045Z","iopub.status.idle":"2022-07-03T19:31:37.904972Z","shell.execute_reply.started":"2022-07-03T19:31:37.897992Z","shell.execute_reply":"2022-07-03T19:31:37.90394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# missing_vals = train_df.isnull().sum().reset_index()\n# missing_vals.columns = ['Column','Count']\n# missing_vals['Missing %'] = np.round(missing_vals['Count'] / train_df.shape[0] * 100,2)\n# missing_vals.sort_values(by='Missing %', ascending=False)\n\n# high_miss_cols = missing_vals[missing_vals['Missing %'] > 95]['Column'].values.tolist()\n# train_df = train_df.drop(high_miss_cols, axis=1)","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-03T19:31:37.906874Z","iopub.execute_input":"2022-07-03T19:31:37.907582Z","iopub.status.idle":"2022-07-03T19:31:37.923252Z","shell.execute_reply.started":"2022-07-03T19:31:37.907545Z","shell.execute_reply":"2022-07-03T19:31:37.921848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Generating aggregated features","metadata":{}},{"cell_type":"code","source":"def agg_feat_eng(df, cols, stat='mean'):\n    feature_names = [f'{feat}_{stat}' for feat in cols]\n    stat_feats_agg = df.groupby('customer_ID')[cols].agg(stat)\n    stat_feats_agg.columns = feature_names\n    return stat_feats_agg","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:37.926181Z","iopub.execute_input":"2022-07-03T19:31:37.926574Z","iopub.status.idle":"2022-07-03T19:31:37.940938Z","shell.execute_reply.started":"2022-07-03T19:31:37.92654Z","shell.execute_reply":"2022-07-03T19:31:37.940011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_df = agg_feat_eng(train_df,mean_cols, stat='mean')\nstd_df = agg_feat_eng(train_df, std_cols, stat='std')\ntrain_data = train_df.groupby('customer_ID').tail(1).set_index('customer_ID')\n\ndel train_df\ngc.collect()\n\ntrain_data = pd.concat([train_data,mean_df,std_df], axis=1)\ntrain_data = train_data[feats_list+['target']]\nprint(f\"Shape of data after adding aggregated features: {train_data.shape}\")\n\ndel mean_df, std_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:31:37.943681Z","iopub.execute_input":"2022-07-03T19:31:37.944313Z","iopub.status.idle":"2022-07-03T19:32:09.681537Z","shell.execute_reply.started":"2022-07-03T19:31:37.944265Z","shell.execute_reply":"2022-07-03T19:32:09.680517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_data['target'] \nX = train_data.drop('target', axis=1)\n\ndel train_data\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:32:09.684194Z","iopub.execute_input":"2022-07-03T19:32:09.684876Z","iopub.status.idle":"2022-07-03T19:32:10.180653Z","shell.execute_reply.started":"2022-07-03T19:32:09.684818Z","shell.execute_reply":"2022-07-03T19:32:10.179758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading and transforming test","metadata":{}},{"cell_type":"code","source":"%%time\ntest = pd.read_feather('../input/amexfeather/test_data.ftr',columns=cols_to_load + ['customer_ID'])","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:32:10.182133Z","iopub.execute_input":"2022-07-03T19:32:10.182497Z","iopub.status.idle":"2022-07-03T19:32:48.331481Z","shell.execute_reply.started":"2022-07-03T19:32:10.182465Z","shell.execute_reply":"2022-07-03T19:32:48.330086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_df = agg_feat_eng(test, mean_cols, stat='mean')\nstd_df = agg_feat_eng(test, std_cols, stat='std')\ntest_df = test.groupby('customer_ID').tail(1).set_index('customer_ID', drop=True).sort_index()\n\ndel test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:32:48.333479Z","iopub.execute_input":"2022-07-03T19:32:48.334053Z","iopub.status.idle":"2022-07-03T19:33:39.294129Z","shell.execute_reply.started":"2022-07-03T19:32:48.334003Z","shell.execute_reply":"2022-07-03T19:33:39.292902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.concat([test_df,mean_df,std_df], axis=1)\ntest_df = test_df[feats_list]\nprint(f\"Shape of data after adding aggregated features: {test_df.shape}\")\n\ndel mean_df, std_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:33:39.29562Z","iopub.execute_input":"2022-07-03T19:33:39.296487Z","iopub.status.idle":"2022-07-03T19:33:41.82109Z","shell.execute_reply.started":"2022-07-03T19:33:39.29645Z","shell.execute_reply":"2022-07-03T19:33:41.819808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:33:41.823799Z","iopub.execute_input":"2022-07-03T19:33:41.824391Z","iopub.status.idle":"2022-07-03T19:33:41.834234Z","shell.execute_reply.started":"2022-07-03T19:33:41.824354Z","shell.execute_reply":"2022-07-03T19:33:41.833111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stratified K-Fold Model Training (with conservative weights added to positive samples)","metadata":{}},{"cell_type":"code","source":"%%time\ngbm_test_preds =[]\nsk_fold = StratifiedKFold(n_splits=10, shuffle=True, random_state=42)\n\nfor fold, (train_idx, val_idx) in enumerate(sk_fold.split(X, y)):\n    \n    print(\"\\nFold {}\".format(fold+1))\n    X_train, y_train = X.iloc[train_idx,:], y[train_idx]\n    X_val, y_val = X.iloc[val_idx,:], y[val_idx]\n    print(\"Train shape: {}, {}, Valid shape: {}, {}\\n\".format(\n        X_train.shape, y_train.shape, X_val.shape, y_val.shape))\n    \n    target_value_counts = y_train.value_counts()\n    scale_pos_weight= np.power(target_value_counts[0] / target_value_counts[1], 1/4)\n    \n    params = {'boosting_type': 'gbdt',\n              'n_estimators': 10000,\n              'num_leaves': 100,\n              'learning_rate': 0.05,\n              'feature_fraction': 0.7,\n              'bagging_fraction': 0.85,\n              'scale_pos_weight':scale_pos_weight,\n              'max_depth':8,\n              'objective': 'binary',\n              'random_state': 42}\n    \n    gbm = LGBMClassifier(**params).fit(X_train, y_train,\n                                       eval_set=[(X_train, y_train), (X_val, y_val)],\n                                       callbacks=[early_stopping(50), log_evaluation(500)],\n                                       eval_metric=['auc','binary_logloss'])\n    gbm_test_preds.append(gbm.predict_proba(test_df)[:,1])\n    \n    print(f\"Fold Score:{np.round(amex_metric(y_val, gbm.predict_proba(X_val)[:,1]),4)}\")\n    \n    del X_train, y_train, X_val, y_val\n    _ = gc.collect()\n    \ndel X,y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T19:33:41.836346Z","iopub.execute_input":"2022-07-03T19:33:41.836868Z","iopub.status.idle":"2022-07-03T19:37:37.694765Z","shell.execute_reply.started":"2022-07-03T19:33:41.836821Z","shell.execute_reply":"2022-07-03T19:37:37.69352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsub['prediction']=np.mean(gbm_test_preds, axis=0)\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T19:48:20.605703Z","iopub.execute_input":"2022-07-02T19:48:20.605998Z","iopub.status.idle":"2022-07-02T19:48:28.286684Z","shell.execute_reply.started":"2022-07-02T19:48:20.605972Z","shell.execute_reply":"2022-07-02T19:48:28.285109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2022-07-02T19:48:28.288797Z","iopub.execute_input":"2022-07-02T19:48:28.289482Z","iopub.status.idle":"2022-07-02T19:48:28.315961Z","shell.execute_reply.started":"2022-07-02T19:48:28.289437Z","shell.execute_reply":"2022-07-02T19:48:28.315067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}