{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# LGBMClassifier!!!","metadata":{}},{"cell_type":"markdown","source":"***procedure***\n\n* Feather dataset used insteas of original data. Well original data is to big!!\n* Deleted date variable and encoded categorical variables.\n* LGBMClassifier is modeled for the data and obatined amex metric.\n* Predicted the results on test data.\n\n***what_next***\n\n* Feature importance have to be done. its a high dimenssional data.\n* The data have to be scaled before modelling.\n* Hyper Parameter tunning is required. optuna is a good one.\n* Startified K-fold must be done because data is imbalanced.\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport optuna  # pip install optuna\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import StratifiedKFold\nimport gc,warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:28:33.408397Z","iopub.execute_input":"2022-08-18T16:28:33.409220Z","iopub.status.idle":"2022-08-18T16:28:35.740546Z","shell.execute_reply.started":"2022-08-18T16:28:33.409097Z","shell.execute_reply":"2022-08-18T16:28:35.739297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Amex Train Data\ndf=pd.read_parquet('../input/amex-parquet/train_data.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:28:35.742914Z","iopub.execute_input":"2022-08-18T16:28:35.743910Z","iopub.status.idle":"2022-08-18T16:29:22.870523Z","shell.execute_reply.started":"2022-08-18T16:28:35.743856Z","shell.execute_reply":"2022-08-18T16:29:22.869086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:21.159644Z","iopub.execute_input":"2022-08-14T06:01:21.160336Z","iopub.status.idle":"2022-08-14T06:01:21.165100Z","shell.execute_reply.started":"2022-08-14T06:01:21.160300Z","shell.execute_reply":"2022-08-14T06:01:21.163796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The train dataset contains customer transactions related to the years of  2017 and 2018.\n#df.sample(n=5,random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:21.168284Z","iopub.execute_input":"2022-08-14T06:01:21.169726Z","iopub.status.idle":"2022-08-14T06:01:21.181264Z","shell.execute_reply.started":"2022-08-14T06:01:21.169672Z","shell.execute_reply":"2022-08-14T06:01:21.180045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the first and last transaction done by the customers.\n\n#f_t=df['S_2']=sort_values(ascending = True,as_index=False).head(1)\n#l_t=df['S_2'].sort_values(ascending = True).tail(1)\n\n#f_t=df['S_2'].min()\n#l_t=df['S_2'].max()\n\n\n#print(('In the given train dataset, first transaction is done on {} and last transaction is on {}.').format(f_t,l_t))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:21.184062Z","iopub.execute_input":"2022-08-14T06:01:21.184803Z","iopub.status.idle":"2022-08-14T06:01:21.194172Z","shell.execute_reply.started":"2022-08-14T06:01:21.184765Z","shell.execute_reply":"2022-08-14T06:01:21.193044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Amex Test Data\n#df_test=pd.read_feather('../input/amexfeather/test_data.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:21.195794Z","iopub.execute_input":"2022-08-14T06:01:21.196880Z","iopub.status.idle":"2022-08-14T06:01:21.205706Z","shell.execute_reply.started":"2022-08-14T06:01:21.196834Z","shell.execute_reply":"2022-08-14T06:01:21.204698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The test dataset contains customer transactions related to the years of  2018 and 2019.\n#f_t=df_test['S_2'].min()\n#l_t=df_test['S_2'].max()\n#print(('In the given test dataset, first transaction is done on {} and last transaction is on {}.').format(f_t,l_t))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:21.208306Z","iopub.execute_input":"2022-08-14T06:01:21.209370Z","iopub.status.idle":"2022-08-14T06:01:21.218318Z","shell.execute_reply.started":"2022-08-14T06:01:21.209331Z","shell.execute_reply":"2022-08-14T06:01:21.216938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":">  The datasets contains huge number of rows and loading the entire dataset in CPU makes the system down.\n\n>  The better way to load is by doing chunks or by using parallel processing like dask.\n\n>  The dataset have an adavantage of reapeated customer transactions and we can get latest transactions  by grouping data.","metadata":{}},{"cell_type":"code","source":"#Grouping by Customer_ID and selecting only the last transaction done by customer.\n\ndf=df.groupby('customer_ID').tail(1)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:29:33.959628Z","iopub.execute_input":"2022-08-18T16:29:33.960193Z","iopub.status.idle":"2022-08-18T16:29:36.679696Z","shell.execute_reply.started":"2022-08-18T16:29:33.960152Z","shell.execute_reply":"2022-08-18T16:29:36.678331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_=gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:29:40.418859Z","iopub.execute_input":"2022-08-18T16:29:40.419320Z","iopub.status.idle":"2022-08-18T16:29:40.725251Z","shell.execute_reply.started":"2022-08-18T16:29:40.419287Z","shell.execute_reply":"2022-08-18T16:29:40.723510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Droping customer_Id and (S_2) date columns.\ndf=df.drop(columns=['S_2'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:29:52.551248Z","iopub.execute_input":"2022-08-18T16:29:52.552208Z","iopub.status.idle":"2022-08-18T16:29:52.768208Z","shell.execute_reply.started":"2022-08-18T16:29:52.552155Z","shell.execute_reply":"2022-08-18T16:29:52.766967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df['target']\nX= df.drop('target',axis =1)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:29:55.690173Z","iopub.execute_input":"2022-08-18T16:29:55.691147Z","iopub.status.idle":"2022-08-18T16:29:55.827510Z","shell.execute_reply.started":"2022-08-18T16:29:55.691096Z","shell.execute_reply":"2022-08-18T16:29:55.826139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=X.drop('customer_ID',axis =1)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:31:47.904098Z","iopub.execute_input":"2022-08-18T16:31:47.904524Z","iopub.status.idle":"2022-08-18T16:31:48.080506Z","shell.execute_reply.started":"2022-08-18T16:31:47.904491Z","shell.execute_reply":"2022-08-18T16:31:48.079012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:31:52.811542Z","iopub.execute_input":"2022-08-18T16:31:52.812017Z","iopub.status.idle":"2022-08-18T16:31:52.957335Z","shell.execute_reply.started":"2022-08-18T16:31:52.811980Z","shell.execute_reply":"2022-08-18T16:31:52.955709Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols=X.select_dtypes(include='object').columns.to_list()\ncat_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:32:05.098228Z","iopub.execute_input":"2022-08-18T16:32:05.098670Z","iopub.status.idle":"2022-08-18T16:32:05.127499Z","shell.execute_reply.started":"2022-08-18T16:32:05.098637Z","shell.execute_reply":"2022-08-18T16:32:05.125921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_col=[col for col in X.columns if col not in cat_cols]","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:33:37.346517Z","iopub.execute_input":"2022-08-18T16:33:37.347814Z","iopub.status.idle":"2022-08-18T16:33:37.359199Z","shell.execute_reply.started":"2022-08-18T16:33:37.347732Z","shell.execute_reply":"2022-08-18T16:33:37.357684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Target is to identify columns having missing more than 95 %..\n# The columns having more missing values add noise to the model\nnull_percent=X[num_col].isna().sum()/len(X[num_col])*100\nnull_f_P=pd.DataFrame({'Features':num_col,'Percentage missing':null_percent})\nnull_f_P=null_f_P.reset_index(drop=True)\nnull_f_P=null_f_P.sort_values('Percentage missing',ascending = False)\nremove_cols=null_f_P[null_f_P['Percentage missing']>98]['Features']\nremove_cols=remove_cols.to_list()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:33:39.648119Z","iopub.execute_input":"2022-08-18T16:33:39.648606Z","iopub.status.idle":"2022-08-18T16:33:40.284906Z","shell.execute_reply.started":"2022-08-18T16:33:39.648569Z","shell.execute_reply":"2022-08-18T16:33:40.283101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:34:06.057653Z","iopub.execute_input":"2022-08-18T16:34:06.058298Z","iopub.status.idle":"2022-08-18T16:34:06.211483Z","shell.execute_reply.started":"2022-08-18T16:34:06.058238Z","shell.execute_reply":"2022-08-18T16:34:06.210253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=X.drop(columns=remove_cols,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:34:19.838774Z","iopub.execute_input":"2022-08-18T16:34:19.839366Z","iopub.status.idle":"2022-08-18T16:34:20.045558Z","shell.execute_reply.started":"2022-08-18T16:34:19.839322Z","shell.execute_reply":"2022-08-18T16:34:20.043951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:34:21.504573Z","iopub.execute_input":"2022-08-18T16:34:21.506411Z","iopub.status.idle":"2022-08-18T16:34:21.662219Z","shell.execute_reply.started":"2022-08-18T16:34:21.506332Z","shell.execute_reply":"2022-08-18T16:34:21.660418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:24.927634Z","iopub.execute_input":"2022-08-14T06:01:24.928248Z","iopub.status.idle":"2022-08-14T06:01:24.933273Z","shell.execute_reply.started":"2022-08-14T06:01:24.928211Z","shell.execute_reply":"2022-08-14T06:01:24.931767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:34:34.071175Z","iopub.execute_input":"2022-08-18T16:34:34.071643Z","iopub.status.idle":"2022-08-18T16:34:34.107034Z","shell.execute_reply.started":"2022-08-18T16:34:34.071610Z","shell.execute_reply":"2022-08-18T16:34:34.105697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Well Machine learing model requires numeric inputs.\n# So idea is to convert categorical variables into numeric type.\n# check here!! https://pandas.pydata.org/docs/reference/api/pandas.get_dummies.html\n\nD_63=pd.get_dummies(X['D_63'])\nX=pd.merge(X,D_63,left_index=True,right_index=True)\nD_64=pd.get_dummies(X['D_64'])\nX=pd.merge(X,D_64,left_index=True,right_index=True)\nX=X.drop(columns=['D_63','D_64'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:42:18.267344Z","iopub.execute_input":"2022-08-18T16:42:18.268349Z","iopub.status.idle":"2022-08-18T16:42:19.733361Z","shell.execute_reply.started":"2022-08-18T16:42:18.268309Z","shell.execute_reply":"2022-08-18T16:42:19.731933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Selecting features which are actually having varaince.\n# The feature selection technique reduces unnecessary noise in the data.\n# The threshold value dependeps on final accuarcy requirements, try hit and trail.\n# check here!! https://scikit-learn.org/stable/modules/feature_selection.html\n\nfrom sklearn.feature_selection import VarianceThreshold\n\nselector = VarianceThreshold(threshold=0.03)\n_=selector.fit(X)\nmask=selector.get_support()\n\n\nther=pd.DataFrame({'columns':X.columns,'thershold':mask})\n\nimp_cols=ther.loc[ther['thershold']==True]\n\ncolumns_to_load=imp_cols['columns'].to_list()\nlen(columns_to_load)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:46:09.521521Z","iopub.execute_input":"2022-08-18T16:46:09.522039Z","iopub.status.idle":"2022-08-18T16:46:11.452489Z","shell.execute_reply.started":"2022-08-18T16:46:09.522001Z","shell.execute_reply":"2022-08-18T16:46:11.450966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Intially the model resulted in poor accuracy because of not using SKFold.\n# given data is imbalanced, so simply going for train-test-split is not a good idea.\n\n#from sklearn.model_selection import train_test_split\n# x_train,x_test,y_train,y_test=train_test_split(X,y,test_size=0.3,random_state=4222)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:27.362346Z","iopub.execute_input":"2022-08-14T06:01:27.362802Z","iopub.status.idle":"2022-08-14T06:01:27.373998Z","shell.execute_reply.started":"2022-08-14T06:01:27.362768Z","shell.execute_reply":"2022-08-14T06:01:27.372877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgbm","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:55:38.751515Z","iopub.execute_input":"2022-08-18T16:55:38.752035Z","iopub.status.idle":"2022-08-18T16:55:40.222728Z","shell.execute_reply.started":"2022-08-18T16:55:38.751996Z","shell.execute_reply":"2022-08-18T16:55:40.221393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_=gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:55:48.257060Z","iopub.execute_input":"2022-08-18T16:55:48.257813Z","iopub.status.idle":"2022-08-18T16:55:48.610688Z","shell.execute_reply.started":"2022-08-18T16:55:48.257775Z","shell.execute_reply":"2022-08-18T16:55:48.608753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"HYPER PARAMETER TUNNING USING OPTUNA!\n> The parameters actaully help in redcucing overfitting and improve accuracy.\n\n> Manually choosing different combinations of parameters is a hard task.\n\n> Best way to select parameters is by using methods like gridsearch or optuna.","metadata":{}},{"cell_type":"code","source":"# check here!! https://optuna.readthedocs.io/en/stable/reference/integration.html#lightgbm\n\nfrom optuna.integration import LightGBMPruningCallback\n\n\ndef objective(trial, X, y):\n    param_grid = {\n        \n        \n        \"objective\": \"binary\",\n        \"metric\": \"auc\",\n        \"boosting_type\": \"gbdt\",\n        'learning_rate': trial.suggest_uniform('learning_rate', 0.001, 0.1),\n        \"lambda_l1\": trial.suggest_loguniform(\"lambda_l1\", 1, 10.0),\n        \"lambda_l2\": trial.suggest_loguniform(\"lambda_l2\", 1, 10.0),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 100),\n        \"colsample_bytree\": trial.suggest_uniform( \"colsample_bytree\", 0.1, 1),\n        \"max_bins\": trial.suggest_int(\"max_bins\", 100, 500),\n        \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 1500, 2500)\n    }\n         \n    \n    cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=4222)\n\n    cv_scores = np.empty(5)\n    for idx, (train_idx, test_idx) in enumerate(cv.split(X, y)):\n        X_train, X_test = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_test = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = lgbm.LGBMClassifier( **param_grid)\n        model.fit(\n            X_train,\n            y_train,\n            eval_set=[(X_test, y_test)],\n            eval_metric='binary_logloss',\n            early_stopping_rounds=200,\n            callbacks=[LightGBMPruningCallback(trial,'binary_logloss')])\n                       \n        preds = model.predict_proba(X_test)\n        cv_scores[idx] = log_loss(y_test, preds)\n\n    return np.mean(cv_scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:56:14.275615Z","iopub.execute_input":"2022-08-18T16:56:14.276073Z","iopub.status.idle":"2022-08-18T16:56:14.299318Z","shell.execute_reply.started":"2022-08-18T16:56:14.276038Z","shell.execute_reply":"2022-08-18T16:56:14.298090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction=\"minimize\", study_name=\"LGBM Classifier\")\nfunc = lambda trial: objective(trial, X, y)\nstudy.optimize(func, n_trials=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T16:56:17.012318Z","iopub.execute_input":"2022-08-18T16:56:17.012772Z","iopub.status.idle":"2022-08-18T17:13:18.068982Z","shell.execute_reply.started":"2022-08-18T16:56:17.012738Z","shell.execute_reply":"2022-08-18T17:13:18.067415Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"\\tBest params:\")\n\nfor key, value in study.best_params.items():\n    print(f\"\\t\\t{key}: {value}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:13:18.074492Z","iopub.execute_input":"2022-08-18T17:13:18.077198Z","iopub.status.idle":"2022-08-18T17:13:18.087019Z","shell.execute_reply.started":"2022-08-18T17:13:18.077143Z","shell.execute_reply":"2022-08-18T17:13:18.084950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid = {\n        \n        'boosting_type': 'gbdt',\n        \"objective\": \"binary\",\n        \"metric\": \"auc\",\n        'learning_rate': 0.0843,\n        \"lambda_l1\": 1.464,\n        \"lambda_l2\": 9.214,\n        \"num_leaves\": 90,\n        \"colsample_bytree\": 0.233,\n        \"max_bins\": 232,\n        \"min_child_samples\": 2165\n    }","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:23:43.576385Z","iopub.execute_input":"2022-08-18T17:23:43.576794Z","iopub.status.idle":"2022-08-18T17:23:43.583733Z","shell.execute_reply.started":"2022-08-18T17:23:43.576764Z","shell.execute_reply":"2022-08-18T17:23:43.582630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" model = lgbm.LGBMClassifier( **param_grid)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:23:45.510409Z","iopub.execute_input":"2022-08-18T17:23:45.511132Z","iopub.status.idle":"2022-08-18T17:23:45.515968Z","shell.execute_reply.started":"2022-08-18T17:23:45.511095Z","shell.execute_reply":"2022-08-18T17:23:45.514995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=4222)\nfor train_idx, test_idx in (cv.split(X, y)):\n    X_train, X_test = X.iloc[train_idx], X.iloc[test_idx]\n    y_train, y_test = y.iloc[train_idx], y.iloc[test_idx]\n    \n    model = lgbm.LGBMClassifier( **param_grid)\n    model.fit(\n            X_train,\n            y_train,\n            eval_set=[(X_test, y_test)],\n            eval_metric='binary_logloss')","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:26:58.634602Z","iopub.execute_input":"2022-08-18T17:26:58.635113Z","iopub.status.idle":"2022-08-18T17:28:36.813059Z","shell.execute_reply.started":"2022-08-18T17:26:58.635076Z","shell.execute_reply":"2022-08-18T17:28:36.811321Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:32:56.992146Z","iopub.execute_input":"2022-08-18T17:32:56.992640Z","iopub.status.idle":"2022-08-18T17:32:57.002647Z","shell.execute_reply.started":"2022-08-18T17:32:56.992598Z","shell.execute_reply":"2022-08-18T17:32:57.001722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=pd.DataFrame(model.predict(X))","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:43:04.133174Z","iopub.execute_input":"2022-08-18T17:43:04.133787Z","iopub.status.idle":"2022-08-18T17:43:08.116289Z","shell.execute_reply.started":"2022-08-18T17:43:04.133742Z","shell.execute_reply":"2022-08-18T17:43:08.114452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:41:20.962024Z","iopub.execute_input":"2022-08-18T17:41:20.962564Z","iopub.status.idle":"2022-08-18T17:41:20.981560Z","shell.execute_reply.started":"2022-08-18T17:41:20.962522Z","shell.execute_reply":"2022-08-18T17:41:20.979944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true=pd.DataFrame(y)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:40:45.532658Z","iopub.execute_input":"2022-08-18T17:40:45.533516Z","iopub.status.idle":"2022-08-18T17:40:45.541486Z","shell.execute_reply.started":"2022-08-18T17:40:45.533474Z","shell.execute_reply":"2022-08-18T17:40:45.540028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=model.predict_proba(X)[:,1]\n\ny_pred=pd.DataFrame(data={'prediction':y_pred})","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:44:05.244791Z","iopub.execute_input":"2022-08-18T17:44:05.245317Z","iopub.status.idle":"2022-08-18T17:44:08.998731Z","shell.execute_reply.started":"2022-08-18T17:44:05.245281Z","shell.execute_reply":"2022-08-18T17:44:08.997783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=y_pred.rename(columns={0:'prediction'})","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:43:09.687569Z","iopub.execute_input":"2022-08-18T17:43:09.688266Z","iopub.status.idle":"2022-08-18T17:43:09.695516Z","shell.execute_reply.started":"2022-08-18T17:43:09.688232Z","shell.execute_reply":"2022-08-18T17:43:09.694004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:29:18.348152Z","iopub.execute_input":"2022-08-18T17:29:18.348590Z","iopub.status.idle":"2022-08-18T17:29:18.363789Z","shell.execute_reply.started":"2022-08-18T17:29:18.348558Z","shell.execute_reply":"2022-08-18T17:29:18.362342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_true,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T17:44:13.487541Z","iopub.execute_input":"2022-08-18T17:44:13.487975Z","iopub.status.idle":"2022-08-18T17:44:15.045054Z","shell.execute_reply.started":"2022-08-18T17:44:13.487941Z","shell.execute_reply":"2022-08-18T17:44:15.043330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" gbm = LGBMClassifier(param_grid).fit(X,y)\n                                       \n                                       \n#gbm_prob = gbm.predict_proba(X_val)[:,1]\n    #gbm_val_probs.append(gbm_prob)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.041816Z","iopub.status.idle":"2022-08-14T06:01:29.042619Z","shell.execute_reply.started":"2022-08-14T06:01:29.042294Z","shell.execute_reply":"2022-08-14T06:01:29.042323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.fit(x_train,y_train)\n\n#y_pred=model.predict_proba(x_test)[:,1]\n\n#y_pred=pd.DataFrame(data={'prediction':y_pred})\n\n#y_true=pd.DataFrame(data={'target':y_test})","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.044437Z","iopub.status.idle":"2022-08-14T06:01:29.045558Z","shell.execute_reply.started":"2022-08-14T06:01:29.045223Z","shell.execute_reply":"2022-08-14T06:01:29.045253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#amex_metric(y_true,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.053750Z","iopub.status.idle":"2022-08-14T06:01:29.054213Z","shell.execute_reply.started":"2022-08-14T06:01:29.053995Z","shell.execute_reply":"2022-08-14T06:01:29.054017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_test=sc.fit_transform(df_test)\n#df_test=pd.DataFrame(df_test)\n#df_test","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.055980Z","iopub.status.idle":"2022-08-14T06:01:29.056376Z","shell.execute_reply.started":"2022-08-14T06:01:29.056189Z","shell.execute_reply":"2022-08-14T06:01:29.056207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y_pred=model.predict_proba(df_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.058008Z","iopub.status.idle":"2022-08-14T06:01:29.058390Z","shell.execute_reply.started":"2022-08-14T06:01:29.058208Z","shell.execute_reply":"2022-08-14T06:01:29.058226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=gbm.predict(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.059829Z","iopub.status.idle":"2022-08-14T06:01:29.060631Z","shell.execute_reply.started":"2022-08-14T06:01:29.060391Z","shell.execute_reply":"2022-08-14T06:01:29.060434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final['prediction']=y_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.062725Z","iopub.status.idle":"2022-08-14T06:01:29.063526Z","shell.execute_reply.started":"2022-08-14T06:01:29.063284Z","shell.execute_reply":"2022-08-14T06:01:29.063307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del df , X,y","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.064970Z","iopub.status.idle":"2022-08-14T06:01:29.065492Z","shell.execute_reply.started":"2022-08-14T06:01:29.065240Z","shell.execute_reply":"2022-08-14T06:01:29.065260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=df_final[['customer_ID','prediction']]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.067632Z","iopub.status.idle":"2022-08-14T06:01:29.068081Z","shell.execute_reply.started":"2022-08-14T06:01:29.067861Z","shell.execute_reply":"2022-08-14T06:01:29.067880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission6.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:29.069451Z","iopub.status.idle":"2022-08-14T06:01:29.069900Z","shell.execute_reply.started":"2022-08-14T06:01:29.069693Z","shell.execute_reply":"2022-08-14T06:01:29.069712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Standard Scaler reduced accuracy from 0.57 to 0.56............","metadata":{}}]}