{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-17T08:19:06.350860Z","iopub.execute_input":"2022-07-17T08:19:06.351569Z","iopub.status.idle":"2022-07-17T08:19:06.384324Z","shell.execute_reply.started":"2022-07-17T08:19:06.351532Z","shell.execute_reply":"2022-07-17T08:19:06.383123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport numpy as np\nimport os\nimport pandas as pd\nimport pickle\nimport sys\nfrom time import time\nfrom tqdm import tqdm\nimport warnings\nfrom sklearn.exceptions import ConvergenceWarning\n\npd.set_option('display.max_columns', None)\nwarnings.filterwarnings(action=\"ignore\", category=ConvergenceWarning)\nwarnings.filterwarnings(action=\"ignore\", category=UserWarning)\nwarnings.filterwarnings(action=\"ignore\", category=FutureWarning)\nwarnings.filterwarnings(action=\"ignore\", category=RuntimeWarning)\n\n# Utils\nfrom IPython.display import display\nimport lightgbm as lgb\nimport matplotlib.pyplot as plt\nfrom scipy import stats\nimport seaborn as sns\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import RocCurveDisplay\nfrom sklearn.model_selection import GridSearchCV,RandomizedSearchCV\nfrom sklearn.model_selection import ParameterGrid\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import KBinsDiscretizer\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\nfrom statsmodels.graphics.gofplots import qqplot\n\nfrom sklearn.experimental import enable_halving_search_cv \n\nfrom sklearn.naive_bayes import GaussianNB\n\nimport lightgbm as lgbm\n\nfrom sklearn.model_selection import HalvingGridSearchCV\n\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import precision_score\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import make_scorer\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import f1_score\n\n\n# roc curve for logistic regression model with optimal threshold\nfrom numpy import sqrt\nfrom numpy import argmax\nfrom matplotlib import pyplot\n\nfrom sklearn.decomposition import PCA \nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom sklearn.utils import class_weight\n\nimport xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:17:36.498849Z","iopub.execute_input":"2022-07-17T09:17:36.499764Z","iopub.status.idle":"2022-07-17T09:17:36.514807Z","shell.execute_reply.started":"2022-07-17T09:17:36.499727Z","shell.execute_reply":"2022-07-17T09:17:36.513563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data = pd.read_csv('/kaggle/input/santander-customer-transaction-prediction/train.csv')\nData.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:07.027814Z","iopub.execute_input":"2022-07-17T08:19:07.028206Z","iopub.status.idle":"2022-07-17T08:19:15.457001Z","shell.execute_reply.started":"2022-07-17T08:19:07.028172Z","shell.execute_reply":"2022-07-17T08:19:15.455920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Data = Data.drop('ID_code',axis =1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:15.458702Z","iopub.execute_input":"2022-07-17T08:19:15.459022Z","iopub.status.idle":"2022-07-17T08:19:15.601199Z","shell.execute_reply.started":"2022-07-17T08:19:15.458992Z","shell.execute_reply":"2022-07-17T08:19:15.600067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(Data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:15.602678Z","iopub.execute_input":"2022-07-17T08:19:15.603007Z","iopub.status.idle":"2022-07-17T08:19:15.612527Z","shell.execute_reply.started":"2022-07-17T08:19:15.602977Z","shell.execute_reply":"2022-07-17T08:19:15.611464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/santander-customer-transaction-prediction/test.csv').drop('ID_code',axis =1)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:15.614859Z","iopub.execute_input":"2022-07-17T08:19:15.615377Z","iopub.status.idle":"2022-07-17T08:19:24.056586Z","shell.execute_reply.started":"2022-07-17T08:19:15.615335Z","shell.execute_reply":"2022-07-17T08:19:24.055684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:24.057802Z","iopub.execute_input":"2022-07-17T08:19:24.058277Z","iopub.status.idle":"2022-07-17T08:19:24.062339Z","shell.execute_reply.started":"2022-07-17T08:19:24.058247Z","shell.execute_reply":"2022-07-17T08:19:24.061581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"Data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:24.063640Z","iopub.execute_input":"2022-07-17T08:19:24.064090Z","iopub.status.idle":"2022-07-17T08:19:26.486664Z","shell.execute_reply.started":"2022-07-17T08:19:24.064062Z","shell.execute_reply":"2022-07-17T08:19:26.485547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:26.488010Z","iopub.execute_input":"2022-07-17T08:19:26.488344Z","iopub.status.idle":"2022-07-17T08:19:28.909532Z","shell.execute_reply.started":"2022-07-17T08:19:26.488315Z","shell.execute_reply":"2022-07-17T08:19:28.908361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The statistics for train and test dataset is quite similar for all variables**","metadata":{}},{"cell_type":"markdown","source":"### Missing Values in Data","metadata":{}},{"cell_type":"code","source":"missing = Data.isnull().sum()\nmissing[missing > 0]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:28.911262Z","iopub.execute_input":"2022-07-17T08:19:28.911960Z","iopub.status.idle":"2022-07-17T08:19:29.010521Z","shell.execute_reply.started":"2022-07-17T08:19:28.911907Z","shell.execute_reply":"2022-07-17T08:19:29.009231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are no missing variables in the data","metadata":{}},{"cell_type":"code","source":"alpha = 1e-3\nall_normal = True\nfor feature in tqdm(Data.columns):\n    if stats.normaltest(Data[feature].values).pvalue > alpha:\n        all_normal = False\n        print(f'{feature} may not be normal')\nif all_normal:\n    print('All features are normally distributed')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:29.012311Z","iopub.execute_input":"2022-07-17T08:19:29.012997Z","iopub.status.idle":"2022-07-17T08:19:29.889037Z","shell.execute_reply.started":"2022-07-17T08:19:29.012950Z","shell.execute_reply":"2022-07-17T08:19:29.887894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The test for normality shows that all variables are normally distributed","metadata":{}},{"cell_type":"markdown","source":"**Let's visualize to understand more about the distribution of the features**","metadata":{}},{"cell_type":"code","source":"def plot_feature_distribution(df1, df2, label1, label2, features):\n    i = 0\n    sns.set_style('whitegrid')\n    plt.figure()\n    fig, ax = plt.subplots(10,10,figsize=(18,22))\n\n    for feature in features:\n        i += 1\n        plt.subplot(10,10,i)\n        sns.distplot(df1[feature], hist=False,label=label1)\n        sns.distplot(df2[feature], hist=False,label=label2)\n        plt.xlabel(feature, fontsize=9)\n        locs, labels = plt.xticks()\n        plt.tick_params(axis='x', which='major', labelsize=6, pad=-6)\n        plt.tick_params(axis='y', which='major', labelsize=6)\n    plt.show();","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:53:57.355710Z","iopub.execute_input":"2022-07-17T08:53:57.356248Z","iopub.status.idle":"2022-07-17T08:53:57.367460Z","shell.execute_reply.started":"2022-07-17T08:53:57.356212Z","shell.execute_reply":"2022-07-17T08:53:57.366616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t0 = Data.loc[Data['target'] == 0]\nt1 = Data.loc[Data['target'] == 1]\nfeatures = Data.columns.values[1:101]\nplot_feature_distribution(t0, t1, '0', '1', features)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:54:58.933434Z","iopub.execute_input":"2022-07-17T08:54:58.933877Z","iopub.status.idle":"2022-07-17T08:56:40.200827Z","shell.execute_reply.started":"2022-07-17T08:54:58.933841Z","shell.execute_reply":"2022-07-17T08:56:40.199428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = Data.columns.values[101:]\nplot_feature_distribution(t0, t1, '0', '1', features)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:56:40.202824Z","iopub.execute_input":"2022-07-17T08:56:40.203166Z","iopub.status.idle":"2022-07-17T08:58:21.506535Z","shell.execute_reply.started":"2022-07-17T08:56:40.203138Z","shell.execute_reply":"2022-07-17T08:58:21.505331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = Data.loc[:,Data.columns!='target']","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:29.892147Z","iopub.execute_input":"2022-07-17T08:19:29.892614Z","iopub.status.idle":"2022-07-17T08:19:30.009564Z","shell.execute_reply.started":"2022-07-17T08:19:29.892581Z","shell.execute_reply":"2022-07-17T08:19:30.008064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = Data.loc[:,Data.columns=='target']\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:30.010912Z","iopub.execute_input":"2022-07-17T08:19:30.011360Z","iopub.status.idle":"2022-07-17T08:19:30.022115Z","shell.execute_reply.started":"2022-07-17T08:19:30.011318Z","shell.execute_reply":"2022-07-17T08:19:30.021256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Outlier Detection","metadata":{}},{"cell_type":"code","source":"#Tukey's method\ndef tukeys_method(df, variable):\n    #Takes two parameters: dataframe & variable of interest as string\n    q1 = df[variable].quantile(0.25)\n    q3 = df[variable].quantile(0.75)\n    iqr = q3-q1\n    outer_fence = 3*iqr\n    \n    #outer fence lower and upper end\n    outer_fence_le = q1-outer_fence\n    outer_fence_ue = q3+outer_fence\n    \n    outliers_prob = []\n\n    for index, x in enumerate(df[variable]):\n        if x <= outer_fence_le or x >= outer_fence_ue:\n            outliers_prob.append(index)\n\n    return outliers_prob\n \ndef percent_outliers(df):\n    no_outliers = True\n    outlier_col = []\n    sum_outliers = {}\n    for col in df.columns:\n        probable_outliers = tukeys_method(df, col)\n        sum_outliers[col] = len(probable_outliers)\n        if len(probable_outliers)>0:\n            outlier_col.append(col)\n            no_outliers= False\n    if no_outliers:\n        return \"No outliers in Data\"\n    else:\n        return sum_outliers\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:19:59.854027Z","iopub.execute_input":"2022-07-17T08:19:59.854454Z","iopub.status.idle":"2022-07-17T08:19:59.866370Z","shell.execute_reply.started":"2022-07-17T08:19:59.854419Z","shell.execute_reply":"2022-07-17T08:19:59.865565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percent_outliers(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:20:13.732289Z","iopub.execute_input":"2022-07-17T08:20:13.732710Z","iopub.status.idle":"2022-07-17T08:20:38.496313Z","shell.execute_reply.started":"2022-07-17T08:20:13.732674Z","shell.execute_reply":"2022-07-17T08:20:38.495074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Checking the distribution target variable","metadata":{}},{"cell_type":"code","source":"sns.countplot(Data.target);","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:21:36.959288Z","iopub.execute_input":"2022-07-17T08:21:36.960188Z","iopub.status.idle":"2022-07-17T08:21:37.125380Z","shell.execute_reply.started":"2022-07-17T08:21:36.960145Z","shell.execute_reply":"2022-07-17T08:21:37.124528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dist = (Data.target.value_counts()/Data.shape[0]*100).to_dict()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:21:49.734115Z","iopub.execute_input":"2022-07-17T08:21:49.735192Z","iopub.status.idle":"2022-07-17T08:21:49.742234Z","shell.execute_reply.started":"2022-07-17T08:21:49.735153Z","shell.execute_reply":"2022-07-17T08:21:49.741444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = dist.values()\nkeys = dist.keys()\n# declaring exploding pie\nexplode = [0, 0.1]\n  \n# plotting data on chart\nplt.pie(data, labels=keys, \n        explode = explode,autopct='%.0f%%')\n  \n# displaying chart\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:21:58.911462Z","iopub.execute_input":"2022-07-17T08:21:58.912527Z","iopub.status.idle":"2022-07-17T08:21:59.017582Z","shell.execute_reply.started":"2022-07-17T08:21:58.912460Z","shell.execute_reply":"2022-07-17T08:21:59.015956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Only 10% of training data has target 1**","metadata":{}},{"cell_type":"markdown","source":"#### Correlation analysis of the data","metadata":{}},{"cell_type":"code","source":"def correl(X_train):\n    cor = X_train.corr()\n    corrm = np.corrcoef(X_train.transpose())\n    corr = corrm - np.diagflat(corrm.diagonal())\n    print(\"max corr:\",corr.max(), \", min corr: \", corr.min())\n    c1 = cor.stack().sort_values(ascending=False).drop_duplicates()\n    high_cor = c1[c1.values!=1]\n    ## change this value to get more correlation results        \n    thresh = 0.95\n    return high_cor[high_cor>thresh]\ncorrel(Data)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:23:17.482054Z","iopub.execute_input":"2022-07-17T08:23:17.482514Z","iopub.status.idle":"2022-07-17T08:23:39.892349Z","shell.execute_reply.started":"2022-07-17T08:23:17.482463Z","shell.execute_reply":"2022-07-17T08:23:39.891031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The max correlation is 0.0667 and min is -0.0809. Let's dive deeper into the correlations","metadata":{}},{"cell_type":"code","source":"train_cor = Data.drop([\"target\"], axis=1).corr()\ntrain_cor = train_cor.values.flatten()\ntrain_cor = train_cor[train_cor != 1]\nplt.figure(figsize=(15,10))\nsns.distplot(train_cor)\nplt.xlabel(\"Correlation values found in training dataset excluding target\")\nplt.ylabel(\"Density\")\nplt.title(\"Correlation between features\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:33:16.648909Z","iopub.execute_input":"2022-07-17T08:33:16.649741Z","iopub.status.idle":"2022-07-17T08:33:38.856974Z","shell.execute_reply.started":"2022-07-17T08:33:16.649696Z","shell.execute_reply":"2022-07-17T08:33:38.855963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Important thing to note is that these correlations are all very small and the corerelations also follow normal distribution. This is for the correlations between the 200 features and the target which is mostly between +0.05 and -0.05. For the correlations between the features themselves they are also waek being mostly between +0.005 and -0.005. There are NO strong correlation hence, feature reduction either by combining features or dropping features will be difficult.","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, Y, test_size=0.20, random_state=42,stratify=Y)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:38:33.418714Z","iopub.execute_input":"2022-07-17T08:38:33.419166Z","iopub.status.idle":"2022-07-17T08:38:34.930469Z","shell.execute_reply.started":"2022-07-17T08:38:33.419133Z","shell.execute_reply":"2022-07-17T08:38:34.929251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nscaled_data = pd.DataFrame(scaler.fit_transform(X_train),columns=X_train.columns) #scaling the data\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:40:19.217604Z","iopub.execute_input":"2022-07-17T08:40:19.218013Z","iopub.status.idle":"2022-07-17T08:40:19.653629Z","shell.execute_reply.started":"2022-07-17T08:40:19.217982Z","shell.execute_reply":"2022-07-17T08:40:19.652647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA().fit(scaled_data)\nplt.plot(np.cumsum(pca.explained_variance_ratio_))\nplt.xlabel('number of components')\nplt.ylabel('cumulative explained variance');\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:40:28.999463Z","iopub.execute_input":"2022-07-17T08:40:28.999884Z","iopub.status.idle":"2022-07-17T08:40:32.238144Z","shell.execute_reply.started":"2022-07-17T08:40:28.999855Z","shell.execute_reply":"2022-07-17T08:40:32.237359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All 100 variables needed to explain 100% of variance, hence the data must be PCA of some other larger dataset and we are more sure that feature reduction would not be a good approach","metadata":{}},{"cell_type":"markdown","source":"# MODELLING","metadata":{}},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\n# define model\nclf_lr = LogisticRegression(random_state=0,class_weight=\"balanced\").fit(X_train, y_train)\n\npreds_lr = clf_lr.predict(X_test\n                         )\nprint(classification_report(y_test, preds_lr))\n\nprint('AUC of test: ',roc_auc_score(y_test, preds_lr ));\n\nprint(confusion_matrix(y_test,preds_lr))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:01:27.520629Z","iopub.execute_input":"2022-07-17T09:01:27.521013Z","iopub.status.idle":"2022-07-17T09:01:32.939828Z","shell.execute_reply.started":"2022-07-17T09:01:27.520982Z","shell.execute_reply":"2022-07-17T09:01:32.938598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Naive Bayes","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import class_weight\n\nsample = class_weight.compute_sample_weight('balanced', y_train)\nclf_nb = GaussianNB()\nclf_nb.fit(X_train,y_train,sample_weight= sample)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:02:01.265551Z","iopub.execute_input":"2022-07-17T09:02:01.266534Z","iopub.status.idle":"2022-07-17T09:02:02.344330Z","shell.execute_reply.started":"2022-07-17T09:02:01.266457Z","shell.execute_reply":"2022-07-17T09:02:02.343248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_nb = clf_nb.predict(X_test)\nprint(classification_report(y_test, preds_nb))\n\nprint(confusion_matrix(y_test,preds_nb))\n\nprint('AUC of test: ',roc_auc_score(y_test, preds_nb ));","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:02:33.723072Z","iopub.execute_input":"2022-07-17T09:02:33.723560Z","iopub.status.idle":"2022-07-17T09:02:33.944781Z","shell.execute_reply.started":"2022-07-17T09:02:33.723512Z","shell.execute_reply":"2022-07-17T09:02:33.943559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Downsampling","metadata":{}},{"cell_type":"markdown","source":"Since the dataset is highly imbalanced, let's try downsampling and then model the data","metadata":{}},{"cell_type":"code","source":"df_0_downsampled = Data[Data[\"target\"]==0].sample(len(Data[Data[\"target\"]==1]), random_state=42)\ndf_1 = Data[Data[\"target\"]==1]\n\ndf_downsampled = pd.concat([df_1, df_0_downsampled], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:08:26.300425Z","iopub.execute_input":"2022-07-17T12:08:26.301667Z","iopub.status.idle":"2022-07-17T12:08:26.535535Z","shell.execute_reply.started":"2022-07-17T12:08:26.301624Z","shell.execute_reply":"2022-07-17T12:08:26.534224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_downsampled.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:04:07.283227Z","iopub.execute_input":"2022-07-17T09:04:07.283988Z","iopub.status.idle":"2022-07-17T09:04:07.294107Z","shell.execute_reply.started":"2022-07-17T09:04:07.283935Z","shell.execute_reply":"2022-07-17T09:04:07.293259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_downsampled.loc[:,df_downsampled.columns!='target']\nY = df_downsampled.loc[:,df_downsampled.columns=='target']","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:08:28.413576Z","iopub.execute_input":"2022-07-17T12:08:28.413999Z","iopub.status.idle":"2022-07-17T12:08:28.436740Z","shell.execute_reply.started":"2022-07-17T12:08:28.413964Z","shell.execute_reply":"2022-07-17T12:08:28.435464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, Y, test_size=0.20, random_state=42,stratify=Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:08:35.665835Z","iopub.execute_input":"2022-07-17T12:08:35.666231Z","iopub.status.idle":"2022-07-17T12:08:35.958734Z","shell.execute_reply.started":"2022-07-17T12:08:35.666201Z","shell.execute_reply":"2022-07-17T12:08:35.957783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic","metadata":{}},{"cell_type":"code","source":"clf_lr = LogisticRegression(random_state=0,class_weight=\"balanced\").fit(X_train, y_train)\n# clf_lr = LogisticRegression(random_state=0,is_unbalance = True)\npreds_lr = clf_lr.predict(X_test)\nprint(classification_report(y_test, preds_lr))\nprint('AUC of test: ',roc_auc_score(y_test, preds_lr ));\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:05:03.818829Z","iopub.execute_input":"2022-07-17T09:05:03.819360Z","iopub.status.idle":"2022-07-17T09:05:04.902657Z","shell.execute_reply.started":"2022-07-17T09:05:03.819308Z","shell.execute_reply":"2022-07-17T09:05:04.901373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's also visualize the most important features using eli5","metadata":{}},{"cell_type":"code","source":"import eli5\neli5.show_weights(clf_lr, feature_names=clf_lr.feature_names_in_)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:05:59.793246Z","iopub.execute_input":"2022-07-17T09:05:59.793691Z","iopub.status.idle":"2022-07-17T09:06:09.845929Z","shell.execute_reply.started":"2022-07-17T09:05:59.793653Z","shell.execute_reply":"2022-07-17T09:06:09.844764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Naive Bayes","metadata":{}},{"cell_type":"code","source":"clf_nb = GaussianNB()\nclf_nb.fit(X_train,y_train)\npreds_nb = clf_nb.predict(X_test)\nprint(classification_report(y_test, preds_nb))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:06:54.433301Z","iopub.execute_input":"2022-07-17T09:06:54.434338Z","iopub.status.idle":"2022-07-17T09:06:54.592179Z","shell.execute_reply.started":"2022-07-17T09:06:54.434297Z","shell.execute_reply":"2022-07-17T09:06:54.591130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('AUC of test: ',roc_auc_score(y_test, preds_nb ));","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:07:05.549726Z","iopub.execute_input":"2022-07-17T09:07:05.550103Z","iopub.status.idle":"2022-07-17T09:07:05.561723Z","shell.execute_reply.started":"2022-07-17T09:07:05.550074Z","shell.execute_reply":"2022-07-17T09:07:05.560509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thus downsampling slightly improves our metric","metadata":{}},{"cell_type":"markdown","source":"### Random forest","metadata":{}},{"cell_type":"markdown","source":"Since all features are important as dedeuced from PCA, let's try now boosting algorithms and check the performance","metadata":{}},{"cell_type":"code","source":"rf = RandomForestClassifier(max_depth=22, max_features='log2', n_estimators=304)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:09:43.766460Z","iopub.execute_input":"2022-07-17T09:09:43.767579Z","iopub.status.idle":"2022-07-17T09:09:43.772338Z","shell.execute_reply.started":"2022-07-17T09:09:43.767540Z","shell.execute_reply":"2022-07-17T09:09:43.771178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf.fit(X_train,y_train.values.ravel())","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:09:50.609011Z","iopub.execute_input":"2022-07-17T09:09:50.609663Z","iopub.status.idle":"2022-07-17T09:11:28.092413Z","shell.execute_reply.started":"2022-07-17T09:09:50.609618Z","shell.execute_reply":"2022-07-17T09:11:28.091310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_rf = rf.predict(X_test)\nprint(classification_report(y_test, preds_rf))\n\nprint('AUC of test: ',roc_auc_score(y_test, preds_rf ));","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:11:40.318797Z","iopub.execute_input":"2022-07-17T09:11:40.319798Z","iopub.status.idle":"2022-07-17T09:11:41.248661Z","shell.execute_reply.started":"2022-07-17T09:11:40.319758Z","shell.execute_reply":"2022-07-17T09:11:41.247559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's visualize important features from Random Forest","metadata":{}},{"cell_type":"code","source":"def plot_feature_importance(importance,names,model_type):\n\n#Create arrays from feature importance and feature names\n    feature_importance = np.array(importance)\n    feature_names = np.array(names)\n\n    #Create a DataFrame using a Dictionary\n    data={'feature_names':feature_names,'feature_importance':feature_importance}\n    fi_df = pd.DataFrame(data)\n    fi_df = fi_df.loc[0:10,:]\n    #Sort the DataFrame in order decreasing feature importance\n    fi_df.sort_values(by=['feature_importance'], ascending=False,inplace=True)\n\n    #Define size of bar plot\n    plt.figure(figsize=(10,8))\n    #Plot Searborn bar chart\n    sns.barplot(x=fi_df['feature_importance'], y=fi_df['feature_names'])\n    #Add chart labels\n    plt.title(model_type + 'FEATURE IMPORTANCE')\n    plt.xlabel('FEATURE IMPORTANCE')\n    plt.ylabel('FEATURE NAMES')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:12:25.956855Z","iopub.execute_input":"2022-07-17T09:12:25.957314Z","iopub.status.idle":"2022-07-17T09:12:25.965966Z","shell.execute_reply.started":"2022-07-17T09:12:25.957273Z","shell.execute_reply":"2022-07-17T09:12:25.964826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_feature_importance(rf.feature_importances_,X.columns,'RANDOM FOREST')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:12:36.985758Z","iopub.execute_input":"2022-07-17T09:12:36.986648Z","iopub.status.idle":"2022-07-17T09:12:37.359063Z","shell.execute_reply.started":"2022-07-17T09:12:36.986612Z","shell.execute_reply":"2022-07-17T09:12:37.357935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### XGboost","metadata":{}},{"cell_type":"markdown","source":"Trying XG boost classifier with best params found out using hyperopt","metadata":{}},{"cell_type":"code","source":"clf_xgb=xgb.XGBClassifier(colsample_bytree= 0.921,\n gamma= 8.3075,\n max_depth=3,\n min_child_weight=1,\n n_estimators=500,\n reg_alpha=3.0,\n reg_lambda=0.798730004926259)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:44:43.100774Z","iopub.execute_input":"2022-07-17T09:44:43.101715Z","iopub.status.idle":"2022-07-17T09:44:43.106980Z","shell.execute_reply.started":"2022-07-17T09:44:43.101679Z","shell.execute_reply":"2022-07-17T09:44:43.105878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_xgb.fit(X_train, y_train\n       )","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:44:45.467875Z","iopub.execute_input":"2022-07-17T09:44:45.468530Z","iopub.status.idle":"2022-07-17T09:47:05.447292Z","shell.execute_reply.started":"2022-07-17T09:44:45.468479Z","shell.execute_reply":"2022-07-17T09:47:05.446107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = clf_xgb.predict(X_test)\nroc_auc_score(y_test, pred>0.5)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T09:47:48.060507Z","iopub.execute_input":"2022-07-17T09:47:48.061401Z","iopub.status.idle":"2022-07-17T09:47:48.107915Z","shell.execute_reply.started":"2022-07-17T09:47:48.061361Z","shell.execute_reply":"2022-07-17T09:47:48.105915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Light GBM","metadata":{}},{"cell_type":"markdown","source":"Modelling using parameters found out by hyperparameter tuning using verstack's LGBM tuner ","metadata":{}},{"cell_type":"code","source":"# X = Data.loc[:,Data.columns!='target']\n# Y = Data.loc[:,Data.columns=='target']","metadata":{"execution":{"iopub.status.busy":"2022-07-17T11:58:21.759851Z","iopub.execute_input":"2022-07-17T11:58:21.760327Z","iopub.status.idle":"2022-07-17T11:58:22.208520Z","shell.execute_reply.started":"2022-07-17T11:58:21.760276Z","shell.execute_reply":"2022-07-17T11:58:22.207208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lgbm = lgb.LGBMClassifier(\nlearning_rate                    = 0.03,\nnum_leaves                       = 66,\ncolsample_bytree                 = 0.9162213204002109,\nsubsample                        = 0.5909124836035503,\nbagging_freq                     = 1,\nmax_depth                        = -1,\nverbosity                        = -1,\nreg_alpha                        = 5.472429642032198e-06,\nreg_lambda                       = 0.00052821153945323,\nmin_split_gain                   = 0.0,\nzero_as_missing                  = False,\nmax_bin                          = 255,\nmin_data_in_bin                  = 3,\nrandom_state                     = 42,\nnum_classes                      = 1,\nobjective                        = 'binary',\nmetric                           = 'binary_logloss',\nnum_threads                      = 0,\nis_unbalance                     = True,\nmin_sum_hessian_in_leaf          = 0.00541524411940254,\nn_estimators                     = 887)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:08:46.571820Z","iopub.execute_input":"2022-07-17T12:08:46.572226Z","iopub.status.idle":"2022-07-17T12:08:46.580322Z","shell.execute_reply.started":"2022-07-17T12:08:46.572194Z","shell.execute_reply":"2022-07-17T12:08:46.579365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lgbm.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:08:49.144157Z","iopub.execute_input":"2022-07-17T12:08:49.144702Z","iopub.status.idle":"2022-07-17T12:09:16.329105Z","shell.execute_reply.started":"2022-07-17T12:08:49.144657Z","shell.execute_reply":"2022-07-17T12:09:16.328037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model_lgbm.predict(X_test)\nroc_auc_score(y_test, pred>0.5)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:09:24.691196Z","iopub.execute_input":"2022-07-17T12:09:24.692435Z","iopub.status.idle":"2022-07-17T12:09:24.925407Z","shell.execute_reply.started":"2022-07-17T12:09:24.692386Z","shell.execute_reply":"2022-07-17T12:09:24.924326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_imp = pd.DataFrame(sorted(zip(model_lgbm.feature_importances_,X.columns)), columns=['Value','Feature'])\nfeature_imp = feature_imp.iloc[0:20,:]\nplt.figure(figsize=(20, 10))\nsns.barplot(x=\"Value\", y=\"Feature\", data=feature_imp.sort_values(by=\"Value\", ascending=False))\nplt.title('LightGBM Features (avg over folds)')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T12:12:12.128852Z","iopub.execute_input":"2022-07-17T12:12:12.129880Z","iopub.status.idle":"2022-07-17T12:12:12.600206Z","shell.execute_reply.started":"2022-07-17T12:12:12.129840Z","shell.execute_reply":"2022-07-17T12:12:12.599100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"markdown","source":"#### Since the best score is with Naive Bayes, we will submit that ","metadata":{}},{"cell_type":"code","source":"submit = pd.DataFrame()\nsubmit['ID_code'] = pd.read_csv('/kaggle/input/santander-customer-transaction-prediction/test.csv')['ID_code']\npreds_submit = clf_nb.predict(test)\n\nsubmit['target'] = preds_submit\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv(\"submission.csv\",index=False)","metadata":{},"execution_count":null,"outputs":[]}]}