{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AMEX prediction with best Hyperparameters for CATBoost,XGBoost & LGB using Hyperopt","metadata":{}},{"cell_type":"markdown","source":"In this notebook we use HyperOpt to find the best hyper parameters to build  ML model for the most famous ML algorithms:\n1. CatBoost\n2. XGBoost\n3. LGB\n\n\nWe start by importing the data and directly moving to hyperparameter tunning. The EDA and feature Importance on this dataset are explained in the below notebook. In this notebook, the focus is on Hyperparameter tuning.\nhttps://www.kaggle.com/code/nandakishorejoshi/amex-pre-and-post-model-predictive-analysis\n","metadata":{}},{"cell_type":"code","source":"#importing required packages\nimport numpy as np # linear algebra\nimport pandas as pd \n\n# visualization tools\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom matplotlib import pyplot\n\nimport gc\n\nfrom catboost import Pool\nimport catboost as ctb\nfrom catboost import *\nimport lightgbm as lgb\nimport xgboost as xgb\n\n\nimport shap\nfrom sklearn import metrics\nfrom time import time\n\nimport os\nimport sklearn\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split,RandomizedSearchCV, StratifiedKFold\nfrom sklearn.metrics import accuracy_score, f1_score, recall_score, confusion_matrix,mean_absolute_error\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score,confusion_matrix\n\nfrom hyperopt import hp\nfrom hyperopt import fmin, tpe, STATUS_OK, STATUS_FAIL, Trials","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:39:34.216313Z","iopub.execute_input":"2022-08-03T02:39:34.217175Z","iopub.status.idle":"2022-08-03T02:39:34.226153Z","shell.execute_reply.started":"2022-08-03T02:39:34.217135Z","shell.execute_reply":"2022-08-03T02:39:34.224533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get the datasets\ndf_train=pd.read_feather('../input/amexfeather/train_data.ftr')\ndf_test=pd.read_feather('../input/amexfeather/test_data.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:19:39.979562Z","iopub.execute_input":"2022-08-03T01:19:39.979917Z","iopub.status.idle":"2022-08-03T01:20:37.197342Z","shell.execute_reply.started":"2022-08-03T01:19:39.979890Z","shell.execute_reply":"2022-08-03T01:20:37.196280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get the latest records for each customer\ntrain_df=df_train.groupby(['customer_ID']).tail(1)\ntest_df=df_test.groupby(['customer_ID']).tail(1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:20:37.199295Z","iopub.execute_input":"2022-08-03T01:20:37.199668Z","iopub.status.idle":"2022-08-03T01:20:44.098146Z","shell.execute_reply.started":"2022-08-03T01:20:37.199632Z","shell.execute_reply":"2022-08-03T01:20:44.097158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting numerical and categorical column names\nnum_cols = df_train.select_dtypes([np.int64,np.float16]).columns\ncat_cols=['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:20:44.099513Z","iopub.execute_input":"2022-08-03T01:20:44.100122Z","iopub.status.idle":"2022-08-03T01:20:48.257845Z","shell.execute_reply.started":"2022-08-03T01:20:44.100082Z","shell.execute_reply":"2022-08-03T01:20:48.256742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Encoding categorical columns\nlab_enc = LabelEncoder()\nfor cat_feat in cat_cols:\n    train_df[cat_feat] = lab_enc.fit_transform(train_df[cat_feat])\n    test_df[cat_feat] = lab_enc.transform(test_df[cat_feat])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:20:48.260439Z","iopub.execute_input":"2022-08-03T01:20:48.260887Z","iopub.status.idle":"2022-08-03T01:20:49.490399Z","shell.execute_reply.started":"2022-08-03T01:20:48.260846Z","shell.execute_reply":"2022-08-03T01:20:49.489281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#defining X and Y data \nX,Y=train_df.drop(['customer_ID','target','S_2'],axis=1),train_df['target']\n\n\n#making sure we have same columns in test data as in train\ncol=[c for c in X.columns]\ntest_x=test_df[col]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:20:55.400252Z","iopub.execute_input":"2022-08-03T01:20:55.400696Z","iopub.status.idle":"2022-08-03T01:20:56.371266Z","shell.execute_reply.started":"2022-08-03T01:20:55.400663Z","shell.execute_reply":"2022-08-03T01:20:56.370254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_df,df_train,df_test,train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:20:58.565625Z","iopub.execute_input":"2022-08-03T01:20:58.566202Z","iopub.status.idle":"2022-08-03T01:20:58.867506Z","shell.execute_reply.started":"2022-08-03T01:20:58.566168Z","shell.execute_reply":"2022-08-03T01:20:58.866519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting data ready to build model\nX_train, X_test, y_train, y_test = train_test_split(X, Y,\n                                                    test_size=0.2,\n                                                    random_state=42,\n                                                    shuffle=True)\n\n\ncategorical_features_indices=[]\nfor c in cat_cols:\n    a=X_train.columns.get_loc(c)\n    categorical_features_indices.append(a)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:21:04.430343Z","iopub.execute_input":"2022-08-03T01:21:04.430832Z","iopub.status.idle":"2022-08-03T01:21:06.055722Z","shell.execute_reply.started":"2022-08-03T01:21:04.430791Z","shell.execute_reply":"2022-08-03T01:21:06.054643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X,Y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:21:08.468365Z","iopub.execute_input":"2022-08-03T01:21:08.468840Z","iopub.status.idle":"2022-08-03T01:21:08.694141Z","shell.execute_reply.started":"2022-08-03T01:21:08.468801Z","shell.execute_reply":"2022-08-03T01:21:08.693254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"\"\"\nX_train shape: {X_train.shape}\nX_test shape: {X_test.shape}\ny_train shape: {y_train.shape}\ny_test shape: {y_test.shape}\n\"\"\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:21:10.449688Z","iopub.execute_input":"2022-08-03T01:21:10.450420Z","iopub.status.idle":"2022-08-03T01:21:10.457194Z","shell.execute_reply.started":"2022-08-03T01:21:10.450366Z","shell.execute_reply":"2022-08-03T01:21:10.455722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We are using Hyperopt to find best Hyperparameters. HyperOpt is a python library for Bayesian Optimization.Following steps are used to implement Hyperopt for effective hyperparameter tuning \n\n1. Define the range of different parameters - This specifies the different values the parameters can take\n2. Define parameter space for each algorithm - Here we are defining different hyparameters we want to tune for each algorithm\n3. Define model fitting condition and loss function for each algorithm\n\n\nOnce all above steps are completed, we create a class to build different models (CatBoost, XGBoost, LGBM) with the above parameter space. Seperate functions are defined in the class for each models. Calling respective functions by passing respective parameter space will run the iterations to tune the specific models.\n\nAfter running the iterations, hyperopt gives the output of the best parametes as indexes of parameter space. We refer to the parameter range to get the actual value and store it in a dictionary. The values in this dictionary is used to build the final models.","metadata":{}},{"cell_type":"code","source":"#define parameter range\nlearning_rate=np.linspace(0.01,0.1,10)\nmax_depth=np.arange(2, 18, 2)\ncolsample_bylevel=np.arange(0.3, 0.8, 0.1)\niterations=np.arange(50, 1000, 50)\nl2_leaf_reg=np.arange(0,10)\nbagging_temperature=np.arange(0,100,10)\nn_estimators=np.arange(50,500,50)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:35:09.197045Z","iopub.execute_input":"2022-08-03T01:35:09.197405Z","iopub.status.idle":"2022-08-03T01:35:09.206176Z","shell.execute_reply.started":"2022-08-03T01:35:09.197374Z","shell.execute_reply":"2022-08-03T01:35:09.203821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Define parameter space, fit conditions, loss function\n\n# XGB parameters\nxgb_cat_params = {\n    'learning_rate':    hp.choice('learning_rate',    learning_rate),\n    'max_depth':        hp.choice('max_depth',         max_depth),\n    'colsample_bytree': hp.choice('colsample_bytree', colsample_bylevel),\n    'n_estimators':     hp.choice('n_estimators',    n_estimators),\n    'loss_function':       'logloss',\n    'nan_mode':'Min',\n    'task_type':'GPU'\n}\nxgb_fit_params = {\n    'eval_metric': 'logloss',\n    'early_stopping_rounds': 10,\n    'verbose': False\n}\nxgb_para = dict()\nxgb_para['cls_params'] = xgb_cat_params\nxgb_para['fit_params'] = xgb_fit_params\nxgb_para['loss_func' ] = lambda y, pred: np.sqrt(mean_squared_error(y, pred))\n\n\n# LightGBM parameters\nlgb_cat_params = {\n    'learning_rate':    hp.choice('learning_rate',    learning_rate),\n    'max_depth':        hp.choice('max_depth',        max_depth),\n    'colsample_bytree': hp.choice('colsample_bytree', colsample_bylevel),\n    'n_estimators':     hp.choice('n_estimators',    n_estimators),\n    'loss_function':       'CrossEntropy',\n    'nan_mode':'Min'\n}\nlgb_fit_params = {\n    'eval_metric': 'CrossEntropy',\n    'early_stopping_rounds': 10,\n    'verbose': False\n}\nlgb_para = dict()\nlgb_para['cls_params'] = lgb_cat_params\nlgb_para['fit_params'] = lgb_fit_params\nlgb_para['loss_func' ] = lambda y, pred: np.sqrt(mean_squared_error(y, pred))\n\n\n# Catboost parameters\nctb_cat_params = {\n    'learning_rate':     hp.choice('learning_rate',    learning_rate),\n    'max_depth':         hp.choice('max_depth',         max_depth),\n    'iterations':       hp.choice('iterations',            iterations),\n    'loss_function':       'CrossEntropy',\n    'nan_mode':'Min',\n    'task_type':'GPU'\n    \n}\nctb_fit_params = {\n    'early_stopping_rounds': 5,\n    'verbose': False,\n    'cat_features': categorical_features_indices\n}\nctb_para = dict()\nctb_para['cls_params'] = ctb_cat_params\nctb_para['fit_params'] = ctb_fit_params\nctb_para['loss_func' ] = lambda y, pred: mean_absolute_error(y, pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:35:18.731293Z","iopub.execute_input":"2022-08-03T01:35:18.731838Z","iopub.status.idle":"2022-08-03T01:35:18.749765Z","shell.execute_reply.started":"2022-08-03T01:35:18.731804Z","shell.execute_reply":"2022-08-03T01:35:18.748526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#define Hyperopt class\nclass HPOpt(object):\n\n    def __init__(self, x_train, x_test, y_train, y_test):\n        self.x_train = x_train\n        self.x_test  = x_test\n        self.y_train = y_train\n        self.y_test  = y_test\n\n    def process(self, fn_name, space, trials, algo, max_evals):\n        fn = getattr(self, fn_name)\n        try:\n            print('entering fmin')\n            result = fmin(fn=fn, space=space, algo=algo, max_evals=max_evals, trials=trials)\n        except Exception as e:\n            return {'status': STATUS_FAIL,\n                    'exception': str(e)}\n        return result\n\n    def ctb_cls(self, para):\n        cls = ctb.CatBoostClassifier(**para['cls_params'])\n        print('ctb initialized')\n        return self.train_cls(cls, para)\n    \n    def xgb_cls(self, para):\n        cls = xgb.XGBClassifier(**para['cls_params'])\n        print('ctb initialized')\n        return self.train_cls(cls, para)\n    \n    def lgb_cls(self, para):\n        cls = lgb.LGBMClassifier(**para['cls_params'])\n        print('ctb initialized')\n        return self.train_cls(cls, para)\n\n    def train_cls(self, cls, para):\n        print('fitting model')\n        cls.fit(self.x_train, self.y_train,\n                eval_set=[(self.x_test, self.y_test)],\n                **para['fit_params'])\n        print('model fitted')\n        pred = cls.predict(self.x_test)\n        #loss = para['loss_func'](self.y_test, pred)\n        f1=sklearn.metrics.f1_score(self.y_test,pred)\n        f1=f1*(-1)\n        print(f1)\n        return {'loss': f1, 'status': STATUS_OK}","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:35:19.986811Z","iopub.execute_input":"2022-08-03T01:35:19.987922Z","iopub.status.idle":"2022-08-03T01:35:19.998940Z","shell.execute_reply.started":"2022-08-03T01:35:19.987877Z","shell.execute_reply":"2022-08-03T01:35:19.997826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tunning Catboost**","metadata":{}},{"cell_type":"code","source":"#calling the Catboost function by passing catboost parameter space \nobj = HPOpt(X_train, X_test, y_train, y_test)\nctb_opt = obj.process(fn_name='ctb_cls', space=ctb_para, trials=Trials(), algo=tpe.suggest, max_evals=2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:01.673765Z","iopub.execute_input":"2022-08-03T01:36:01.674170Z","iopub.status.idle":"2022-08-03T01:47:13.869514Z","shell.execute_reply.started":"2022-08-03T01:36:01.674137Z","shell.execute_reply":"2022-08-03T01:47:13.868489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output of the hyperopt is the indexes of the best parameters\nctb_opt","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:47:13.871711Z","iopub.execute_input":"2022-08-03T01:47:13.872367Z","iopub.status.idle":"2022-08-03T01:47:13.880025Z","shell.execute_reply.started":"2022-08-03T01:47:13.872328Z","shell.execute_reply":"2022-08-03T01:47:13.878657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save best parametrs in a dictionary\nbest_param_ctb={}\nbest_param_ctb['learning_rate']=learning_rate[ctb_opt['learning_rate']]\nbest_param_ctb['iterations']=iterations[ctb_opt['iterations']]\nbest_param_ctb['max_depth']=max_depth[ctb_opt['max_depth']]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:47:13.881805Z","iopub.execute_input":"2022-08-03T01:47:13.882478Z","iopub.status.idle":"2022-08-03T01:47:13.888437Z","shell.execute_reply.started":"2022-08-03T01:47:13.882441Z","shell.execute_reply":"2022-08-03T01:47:13.887437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dictionary has the actual  hyperparameters\nbest_param_ctb","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:47:13.891126Z","iopub.execute_input":"2022-08-03T01:47:13.892175Z","iopub.status.idle":"2022-08-03T01:47:13.903554Z","shell.execute_reply.started":"2022-08-03T01:47:13.892139Z","shell.execute_reply":"2022-08-03T01:47:13.902583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tuning LGBM**","metadata":{}},{"cell_type":"code","source":"#calling the LGBM function by passing LGBM parameter space \n#obj = HPOpt(X_train, X_test, y_train, y_test)\nlgb_opt = obj.process(fn_name='lgb_cls', space=lgb_para, trials=Trials(), algo=tpe.suggest, max_evals=2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:47:13.905008Z","iopub.execute_input":"2022-08-03T01:47:13.906047Z","iopub.status.idle":"2022-08-03T01:51:42.741831Z","shell.execute_reply.started":"2022-08-03T01:47:13.906009Z","shell.execute_reply":"2022-08-03T01:51:42.740793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output of the hyperopt is the indexes of the best parameters\nlgb_opt","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:51:42.744385Z","iopub.execute_input":"2022-08-03T01:51:42.745097Z","iopub.status.idle":"2022-08-03T01:51:42.752328Z","shell.execute_reply.started":"2022-08-03T01:51:42.745058Z","shell.execute_reply":"2022-08-03T01:51:42.751152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save best parameters in a dictionary\nbest_param_lgb={}\nbest_param_lgb['learning_rate']=learning_rate[lgb_opt['learning_rate']]\nbest_param_lgb['colsample_bytree']=colsample_bylevel[lgb_opt['colsample_bytree']]\nbest_param_lgb['max_depth']=max_depth[lgb_opt['max_depth']]\nbest_param_lgb['n_estimators']=n_estimators[lgb_opt['n_estimators']]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:51:42.753802Z","iopub.execute_input":"2022-08-03T01:51:42.754350Z","iopub.status.idle":"2022-08-03T01:51:42.761024Z","shell.execute_reply.started":"2022-08-03T01:51:42.754312Z","shell.execute_reply":"2022-08-03T01:51:42.760000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dictionary has the actual  hyperparameters\nbest_param_lgb","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:51:42.762801Z","iopub.execute_input":"2022-08-03T01:51:42.763233Z","iopub.status.idle":"2022-08-03T01:51:42.773212Z","shell.execute_reply.started":"2022-08-03T01:51:42.763194Z","shell.execute_reply":"2022-08-03T01:51:42.772144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tuning XGBoost**","metadata":{}},{"cell_type":"code","source":"#calling the XGBoost function by passing XGB parameter space \nxgb_opt = obj.process(fn_name='xgb_cls', space=xgb_para, trials=Trials(), algo=tpe.suggest, max_evals=2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking the output of hyperopt\nxgb_opt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save best parameters in a dictionary\nbest_param_xgb={}\nbest_param_xgb['learning_rate']=learning_rate[xgb_opt['learning_rate']]\nbest_param_xgb['colsample_bytree']=colsample_bylevel[xgb_opt['colsample_bytree']]\nbest_param_xgb['max_depth']=max_depth[xgb_opt['max_depth']]\nbest_param_xgb['n_estimators']=n_estimators[xgb_opt['n_estimators']]\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_param_xgb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building Models**\n\nNow we use the best parameters from  above steps to build models","metadata":{}},{"cell_type":"markdown","source":"**Building LGBM Model**","metadata":{}},{"cell_type":"code","source":"#build model with the best hyperparameters\nmodel_lgb=lgb.LGBMClassifier(n_estimators=best_param_lgb['n_estimators'], \n                             depth=best_param_lgb['max_depth'],\n                             learning_rate=best_param_lgb['learning_rate'],\n                             colsample_bytree=best_param_lgb['colsample_bytree'],\n                            loss_function='logloss',\n                             nan_mode='Min',\n                            #task_type='GPU',\n                             \n                            random_seed=42\n                            )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lgb.fit(X_train,y_train, eval_set=None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building XGBoost Model**","metadata":{}},{"cell_type":"code","source":"#build model with the best hyperparameters\nmodel_xgb=xgb.XGBClassifier(n_estimators=best_param_xgb['n_estimators'], \n                             depth=best_param_xgb['max_depth'],\n                             learning_rate=best_param_xgb['learning_rate'],\n                             colsample_bytree=best_param_xgb['colsample_bytree'],\n                            loss_function='logloss',\n                             nan_mode='Min',\n                            task_type='GPU',\n                             \n                            random_seed=42\n                            )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_xgb.fit(X_train,y_train, eval_set=None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building CatBoost Model**","metadata":{}},{"cell_type":"code","source":"#build model with the best hyperparameters\nmodel_ctb=ctb.CatBoostClassifier(iterations=best_param_ctb['iterations'], \n                             depth=best_param_ctb['max_depth'],\n                             learning_rate=best_param_ctb['learning_rate'],\n                            loss_function='CrossEntropy',\n                             nan_mode='Min',\n                            task_type='GPU',\n                            random_seed=42\n                            )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_ctb.fit(X_train,y_train,cat_features=categorical_features_indices, eval_set=None, plot=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Make predictions on test data to check model fit","metadata":{}},{"cell_type":"code","source":"#Catboost\ny_pred_ctb=model_ctb.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#XGBoost\ny_pred_xgb=model_xgb.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#LGBM\ny_pred_lgb=model_lgb.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('F1-score of Catboost: {:.2f}'.format(f1_score(y_test, y_pred_ctb)))\nprint('F1-score of XGBoost: {:.2f}'.format(f1_score(y_test, y_pred_xgb)))\nprint('F1-score of LGBM: {:.2f}'.format(f1_score(y_test, y_pred_lgb)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submition of the results**\n\nAMEX has custom metric. The below code shows a way to make submission","metadata":{}},{"cell_type":"code","source":"#Amex metric as provided in competition\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:48:53.176123Z","iopub.execute_input":"2022-07-30T01:48:53.176493Z","iopub.status.idle":"2022-07-30T01:48:53.190462Z","shell.execute_reply.started":"2022-07-30T01:48:53.176461Z","shell.execute_reply":"2022-07-30T01:48:53.189302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting series to dataframe with required column names\ny_test1 = pd.DataFrame(y_test, columns=['target'])\ny_pred1 = pd.DataFrame(y_pred_ctb, columns=['prediction'])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T01:48:56.407517Z","iopub.execute_input":"2022-07-30T01:48:56.408439Z","iopub.status.idle":"2022-07-30T01:48:56.417002Z","shell.execute_reply.started":"2022-07-30T01:48:56.408394Z","shell.execute_reply":"2022-07-30T01:48:56.416033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get the metric value\n#print(amex_metric(y_test1, y_pred1))","metadata":{},"execution_count":null,"outputs":[]}]}