{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## **Import Necessary Library**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:20:23.859560Z","iopub.execute_input":"2022-07-15T03:20:23.860563Z","iopub.status.idle":"2022-07-15T03:20:28.515399Z","shell.execute_reply.started":"2022-07-15T03:20:23.860462Z","shell.execute_reply":"2022-07-15T03:20:28.514344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CONFIG:\n    random_state = 4222\n    kaggle = True\n    path = '../input/amexfeather'\n    local_path = ''","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:20:31.540502Z","iopub.execute_input":"2022-07-15T03:20:31.540908Z","iopub.status.idle":"2022-07-15T03:20:31.546035Z","shell.execute_reply.started":"2022-07-15T03:20:31.540875Z","shell.execute_reply":"2022-07-15T03:20:31.545001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"#%%time\n#train = pd.read_feather(CONFIG.path + '/train_data.ftr')\n\ntrain = pd.read_feather(f'{CONFIG.path}/train_data.ftr')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:20:37.662525Z","iopub.execute_input":"2022-07-15T03:20:37.663069Z","iopub.status.idle":"2022-07-15T03:21:01.255732Z","shell.execute_reply.started":"2022-07-15T03:20:37.663033Z","shell.execute_reply":"2022-07-15T03:21:01.254145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No customer falls into default within the 13-month period. \n\n#D = pd.DataFrame(train, columns =[\"customer_ID\", \"target\"])\n#D = D.loc[D[\"target\"] == 1]\n#D.head()\n#del train\n\n#yy = D.groupby(\"customer_ID\", as_index=False).sum()\n#ys = D.groupby(\"customer_ID\",as_index=False).size()\n\n#yy['target'].equals(ys['size'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select last statement per customer and make customer_ID the index\n\ntrain = train.groupby('customer_ID')\ntrain = train.tail(1)\ntrain = train.drop(['S_2'], axis=1)\ntrain.set_index('customer_ID', inplace=True) \ntrain.tail(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:05.175434Z","iopub.execute_input":"2022-07-15T03:21:05.175801Z","iopub.status.idle":"2022-07-15T03:21:07.733085Z","shell.execute_reply.started":"2022-07-15T03:21:05.175774Z","shell.execute_reply":"2022-07-15T03:21:07.732184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"reference: https://www.kaggle.com/competitions/amex-default-prediction/discussion/327094","metadata":{}},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:19.475927Z","iopub.execute_input":"2022-07-15T03:21:19.476337Z","iopub.status.idle":"2022-07-15T03:21:19.616752Z","shell.execute_reply.started":"2022-07-15T03:21:19.476301Z","shell.execute_reply":"2022-07-15T03:21:19.615833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_cols = train.columns.to_list()\ncat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnum_features = [col for col in total_cols if col not in cat_features + [\"target\", \"customer_ID\", \"S_2\"] ]\nlen(num_features) + len(cat_features)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:29.254520Z","iopub.execute_input":"2022-07-15T03:21:29.254887Z","iopub.status.idle":"2022-07-15T03:21:29.260586Z","shell.execute_reply.started":"2022-07-15T03:21:29.254857Z","shell.execute_reply":"2022-07-15T03:21:29.259631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train[cat_features + num_features]\ny = train['target']\n\nx.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:35.965149Z","iopub.execute_input":"2022-07-15T03:21:35.965682Z","iopub.status.idle":"2022-07-15T03:21:36.342893Z","shell.execute_reply.started":"2022-07-15T03:21:35.965637Z","shell.execute_reply":"2022-07-15T03:21:36.342079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Apply OrdinalEncoder","metadata":{"execution":{"iopub.status.busy":"2022-06-24T10:37:48.21323Z","iopub.execute_input":"2022-06-24T10:37:48.213809Z","iopub.status.idle":"2022-06-24T10:37:48.217939Z","shell.execute_reply.started":"2022-06-24T10:37:48.213777Z","shell.execute_reply":"2022-06-24T10:37:48.216658Z"}}},{"cell_type":"code","source":"%%time\n\nenc = OrdinalEncoder()\nx[cat_features] = enc.fit_transform(x[cat_features])\n_ = gc.collect","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:40.030960Z","iopub.execute_input":"2022-07-15T03:21:40.031623Z","iopub.status.idle":"2022-07-15T03:21:41.090896Z","shell.execute_reply.started":"2022-07-15T03:21:40.031586Z","shell.execute_reply":"2022-07-15T03:21:41.090052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Competition Metric**","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:44.222176Z","iopub.execute_input":"2022-07-15T03:21:44.222930Z","iopub.status.idle":"2022-07-15T03:21:44.243393Z","shell.execute_reply.started":"2022-07-15T03:21:44.222881Z","shell.execute_reply":"2022-07-15T03:21:44.242468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model Training**","metadata":{}},{"cell_type":"code","source":"# Next step: implement a K-fold CV method\n\nX_train, X_test, y_train, y_test = train_test_split(x,y,\n                            test_size=0.3,random_state=CONFIG.random_state, \n                                                    stratify = y)\nX_train.shape, X_test.shape, y_train.shape, y_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:47.916361Z","iopub.execute_input":"2022-07-15T03:21:47.916777Z","iopub.status.idle":"2022-07-15T03:21:49.266121Z","shell.execute_reply.started":"2022-07-15T03:21:47.916745Z","shell.execute_reply":"2022-07-15T03:21:49.265198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import optuna\n#import optuna.integration.lightgbm as lgb\n#from sklearn.model_selection import RepeatedKFold\n\n#import warnings\n#warnings.simplefilter(action='ignore', category=FutureWarning)\n#warnings.simplefilter(action='ignore', category=UserWarning)\n\n#rkf = RepeatedKFold(n_splits = 3, n_repeats = 3, random_state=42)\n\n#fixed_params = {\n#    'objective': 'binary',\n#    'metric': 'auc',\n#    'boosting_type': 'gbdt',\n#    'force_row_wise' : True,\n#    'device':'gpu',\n#    'random_state' : CONFIG.random_state,\n#    'extra_trees' : True,\n#    'feature_pre_filter': False,\n#    'verbose' : -1,\n#    'n_estimators': 300,\n#    'early_stopping_round': 30\n#}\n\n#X = np.array(X_train[features])\n#y = np.array(y_train).flatten()\n\n#dtrain = lgb.Dataset(X, label = y, categorical_feature = 'auto')    \n\n#tuner = lgb.LightGBMTunerCV(\n#        fixed_params, dtrain, \n#        verbose_eval = None,\n#        time_budget = 60000,\n#        folds = rkf,\n#        #num_boost_round = 10,\n#        shuffle = True\n#)\n\n#tuner.run()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T02:42:09.234976Z","iopub.execute_input":"2022-07-11T02:42:09.235755Z","iopub.status.idle":"2022-07-11T02:42:09.240411Z","shell.execute_reply.started":"2022-07-11T02:42:09.235719Z","shell.execute_reply":"2022-07-11T02:42:09.239476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tuner.best_params","metadata":{"execution":{"iopub.status.busy":"2022-07-11T01:28:51.267077Z","iopub.execute_input":"2022-07-11T01:28:51.267469Z","iopub.status.idle":"2022-07-11T01:28:51.274076Z","shell.execute_reply.started":"2022-07-11T01:28:51.267435Z","shell.execute_reply":"2022-07-11T01:28:51.273060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# seach parameters from Optuna hypertuning and hand waving on num_leaves\n\nsearch_params = { \n    'learning_rate' : 0.065,\n    'lambda_l1': 5.465073323789261,\n    'lambda_l2': 8.252315889074575,\n    'num_leaves': 220,\n    'feature_fraction': 0.62,\n    'bagging_fraction': 0.9898640166643957,\n    'bagging_freq': 3,\n    'min_child_samples': 100\n}\n\nfixed_params={\n    'objective': 'binary',\n    'metric': 'auc',\n    'boosting_type' : 'gbdt',\n    'force_row_wise' : True,\n    'device': 'gpu',\n    'random_state' : CONFIG.random_state,\n    'extra_trees' : True,\n    'feature_pre_filter': False,\n    'n_estimators': 300,\n    'early_stopping_round': 30\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:55.885776Z","iopub.execute_input":"2022-07-15T03:21:55.886143Z","iopub.status.idle":"2022-07-15T03:21:55.895041Z","shell.execute_reply.started":"2022-07-15T03:21:55.886112Z","shell.execute_reply":"2022-07-15T03:21:55.891899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LGBMClassifier(**fixed_params, **search_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:21:59.507345Z","iopub.execute_input":"2022-07-15T03:21:59.507922Z","iopub.status.idle":"2022-07-15T03:21:59.512474Z","shell.execute_reply.started":"2022-07-15T03:21:59.507888Z","shell.execute_reply":"2022-07-15T03:21:59.511249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel.fit(\n    X_train, y_train, \n    eval_set=[(X_test,y_test)],\n    callbacks=[log_evaluation(100)]\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:22:02.755915Z","iopub.execute_input":"2022-07-15T03:22:02.756455Z","iopub.status.idle":"2022-07-15T03:22:54.764119Z","shell.execute_reply.started":"2022-07-15T03:22:02.756420Z","shell.execute_reply":"2022-07-15T03:22:54.763191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Check trained model w/ X_test, y_test, and amex_metric**","metadata":{}},{"cell_type":"code","source":"%%time\ny_pred = pd.DataFrame(y_test.copy(deep=True))\ny_pred = y_pred.rename(columns={'target':'prediction'})\n\ny_pred[\"prediction\"] = model.predict_proba(X_test)[:,1]\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:22:58.751979Z","iopub.execute_input":"2022-07-15T03:22:58.752356Z","iopub.status.idle":"2022-07-15T03:23:06.064981Z","shell.execute_reply.started":"2022-07-15T03:22:58.752324Z","shell.execute_reply":"2022-07-15T03:23:06.064269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:23:12.745254Z","iopub.execute_input":"2022-07-15T03:23:12.745948Z","iopub.status.idle":"2022-07-15T03:23:12.751748Z","shell.execute_reply.started":"2022-07-15T03:23:12.745912Z","shell.execute_reply":"2022-07-15T03:23:12.750913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\namex_metric(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:23:15.547456Z","iopub.execute_input":"2022-07-15T03:23:15.548019Z","iopub.status.idle":"2022-07-15T03:23:15.992093Z","shell.execute_reply.started":"2022-07-15T03:23:15.547983Z","shell.execute_reply":"2022-07-15T03:23:15.991298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Confusion matrix**","metadata":{}},{"cell_type":"code","source":"y_pred_actual = model.predict(X_test)\ncm = confusion_matrix(y_test, y_pred_actual)\ncmd = ConfusionMatrixDisplay(cm,display_labels=['Not','Default'])\ncmd.plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T03:23:26.290385Z","iopub.execute_input":"2022-07-15T03:23:26.291054Z","iopub.status.idle":"2022-07-15T03:23:37.283492Z","shell.execute_reply.started":"2022-07-15T03:23:26.291009Z","shell.execute_reply":"2022-07-15T03:23:37.282752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Submission**","metadata":{}},{"cell_type":"code","source":"del train, x, y, X_test, X_train, y_train, y_test, y_pred\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T11:51:08.944158Z","iopub.execute_input":"2022-06-24T11:51:08.944514Z","iopub.status.idle":"2022-06-24T11:51:09.257159Z","shell.execute_reply.started":"2022-06-24T11:51:08.944484Z","shell.execute_reply":"2022-06-24T11:51:09.256356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest = pd.read_feather(f'{config.path}/test_data.ftr')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T11:51:24.407173Z","iopub.execute_input":"2022-06-24T11:51:24.407529Z","iopub.status.idle":"2022-06-24T11:52:09.175296Z","shell.execute_reply.started":"2022-06-24T11:51:24.407499Z","shell.execute_reply":"2022-06-24T11:52:09.174516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.groupby('customer_ID')\ntest = test.tail(1)\ntest = test.drop(['S_2'], axis=1)\ntest.set_index('customer_ID', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T11:52:29.92943Z","iopub.execute_input":"2022-06-24T11:52:29.930092Z","iopub.status.idle":"2022-06-24T11:52:34.696354Z","shell.execute_reply.started":"2022-06-24T11:52:29.930055Z","shell.execute_reply":"2022-06-24T11:52:34.695522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T11:52:36.389364Z","iopub.execute_input":"2022-06-24T11:52:36.389739Z","iopub.status.idle":"2022-06-24T11:52:36.671623Z","shell.execute_reply.started":"2022-06-24T11:52:36.389706Z","shell.execute_reply":"2022-06-24T11:52:36.670649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest[cat_features] = enc.transform(test[cat_features])\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T11:52:55.935944Z","iopub.execute_input":"2022-06-24T11:52:55.936316Z","iopub.status.idle":"2022-06-24T11:52:57.60987Z","shell.execute_reply.started":"2022-06-24T11:52:55.936285Z","shell.execute_reply":"2022-06-24T11:52:57.608682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"prediction\"] = model.predict_proba(test[cat_features + num_features])[:,1]\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T11:53:20.312133Z","iopub.execute_input":"2022-06-24T11:53:20.312496Z","iopub.status.idle":"2022-06-24T11:53:47.84907Z","shell.execute_reply.started":"2022-06-24T11:53:20.312467Z","shell.execute_reply":"2022-06-24T11:53:47.848293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"prediction\"].to_csv(\"submission.csv\", index=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-24T11:53:54.639958Z","iopub.execute_input":"2022-06-24T11:53:54.640387Z","iopub.status.idle":"2022-06-24T11:53:59.01044Z","shell.execute_reply.started":"2022-06-24T11:53:54.640351Z","shell.execute_reply":"2022-06-24T11:53:59.009621Z"},"trusted":true},"execution_count":null,"outputs":[]}]}