{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### 1. Data pre-processing ","metadata":{}},{"cell_type":"code","source":"import os\nos.listdir('../input/amex-default-prediction')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-02T08:28:20.919140Z","iopub.execute_input":"2022-09-02T08:28:20.919716Z","iopub.status.idle":"2022-09-02T08:28:20.951863Z","shell.execute_reply.started":"2022-09-02T08:28:20.919564Z","shell.execute_reply":"2022-09-02T08:28:20.950922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport gc\n\ndf_train = pd.read_parquet(\"../input/train-test-parquet/train.parquet\")\ndf_train_labels = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\n\nprint(len(df_train))\nprint(df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:28:20.953451Z","iopub.execute_input":"2022-09-02T08:28:20.953794Z","iopub.status.idle":"2022-09-02T08:28:43.472822Z","shell.execute_reply.started":"2022-09-02T08:28:20.953765Z","shell.execute_reply":"2022-09-02T08:28:43.470755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Due to large dataset of multiple transaction from individual customers at different transactional date, only latest transaction of each customers has been utilized to determine if customers will or will not default credit cards.","metadata":{}},{"cell_type":"code","source":"# Get only the latest transaction of each customer_ID\ndf_train = df_train.groupby('customer_ID').tail(1)\nprint(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:28:43.476181Z","iopub.execute_input":"2022-09-02T08:28:43.477408Z","iopub.status.idle":"2022-09-02T08:28:48.925459Z","shell.execute_reply.started":"2022-09-02T08:28:43.477370Z","shell.execute_reply":"2022-09-02T08:28:48.922042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Due to individual customers' data exist across majority of the rows, only remove columns that has more than or equal to 50% of missing entries as it would affect the performance of prediction modelling.","metadata":{}},{"cell_type":"code","source":"#((df_train.isnull() | df_train.isna()).sum() * 100 / df_train.index.size).round(5)\n#count_nulls = (df_train.isnull() | df_train.isna()).mean().round(5)*100 # checks for NAs\n#count_nulls = count_nulls.drop(count_nulls[count_nulls <= 50].index)\n#print(count_nulls)\n\ndf_train = df_train.drop(df_train.columns[df_train.apply(lambda col: (col.isnull() | col.isna()).mean().round(5)*100 >= 50)], axis=1)\nprint(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:28:48.933491Z","iopub.execute_input":"2022-09-02T08:28:48.934676Z","iopub.status.idle":"2022-09-02T08:28:49.702533Z","shell.execute_reply.started":"2022-09-02T08:28:48.934525Z","shell.execute_reply":"2022-09-02T08:28:49.701418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.merge(df_train, df_train_labels, on='customer_ID' )","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:28:49.704251Z","iopub.execute_input":"2022-09-02T08:28:49.704977Z","iopub.status.idle":"2022-09-02T08:29:00.449158Z","shell.execute_reply.started":"2022-09-02T08:28:49.704932Z","shell.execute_reply":"2022-09-02T08:29:00.447620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_var = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68', 'target']\ndf_train[cat_var] = df_train[cat_var].fillna(0)\n\nprint(df_train[cat_var].dtypes)\ndf_train[['D_63','D_64']] = df_train[['D_63','D_64']].astype('category').apply(lambda x: x.cat.codes)\nprint(df_train[cat_var].dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:00.450959Z","iopub.execute_input":"2022-09-02T08:29:00.451828Z","iopub.status.idle":"2022-09-02T08:29:00.682482Z","shell.execute_reply.started":"2022-09-02T08:29:00.451791Z","shell.execute_reply":"2022-09-02T08:29:00.681291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().all().sum() # check if entries of all rows are empty","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:00.684173Z","iopub.execute_input":"2022-09-02T08:29:00.684551Z","iopub.status.idle":"2022-09-02T08:29:00.764185Z","shell.execute_reply.started":"2022-09-02T08:29:00.684517Z","shell.execute_reply":"2022-09-02T08:29:00.762968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:00.765701Z","iopub.execute_input":"2022-09-02T08:29:00.766085Z","iopub.status.idle":"2022-09-02T08:29:01.048993Z","shell.execute_reply.started":"2022-09-02T08:29:00.766038Z","shell.execute_reply":"2022-09-02T08:29:01.047837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove redundant variables: customer_ID and S_2 (date).\ndf_train.drop(df_train.iloc[:,0:2], axis=1, inplace=True)\ndf_train = df_train.apply(abs)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:01.050716Z","iopub.execute_input":"2022-09-02T08:29:01.051374Z","iopub.status.idle":"2022-09-02T08:29:02.013001Z","shell.execute_reply.started":"2022-09-02T08:29:01.051328Z","shell.execute_reply":"2022-09-02T08:29:02.011877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_train_labels\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.018143Z","iopub.execute_input":"2022-09-02T08:29:02.018486Z","iopub.status.idle":"2022-09-02T08:29:02.221034Z","shell.execute_reply.started":"2022-09-02T08:29:02.018456Z","shell.execute_reply":"2022-09-02T08:29:02.219651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Visualize the distirbution of defaulters.","metadata":{}},{"cell_type":"code","source":"count_target = df_train.target.value_counts()\nprint(count_target)\nprint(\"\\nPredicting only 0 = {:.2f}% accuracy\".format(count_target[0] / sum(count_target) * 100))\nprint(\"\\nPredicting only 1 = {:.2f}% accuracy\".format(count_target[1] / sum(count_target) * 100))","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.222408Z","iopub.execute_input":"2022-09-02T08:29:02.222768Z","iopub.status.idle":"2022-09-02T08:29:02.237930Z","shell.execute_reply.started":"2022-09-02T08:29:02.222729Z","shell.execute_reply":"2022-09-02T08:29:02.236945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It can be seen above that imbalanced dataset existed as it only contains 25.89% portion of information on customers who default(1) . In contrast, there are a handful of information (74.11%) on customer who does not default their credit cards (0). Thus, it may be prone to lower accuracy on prediction model.","metadata":{}},{"cell_type":"code","source":"df_count_target = pd.DataFrame(count_target).reset_index()\nprint(df_count_target)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.239531Z","iopub.execute_input":"2022-09-02T08:29:02.240140Z","iopub.status.idle":"2022-09-02T08:29:02.255360Z","shell.execute_reply.started":"2022-09-02T08:29:02.240106Z","shell.execute_reply":"2022-09-02T08:29:02.254102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns \nimport matplotlib.pyplot as plt\n\nsns.reset_orig()\nplt.figure(figsize = (5,5))\nmy_palette = sns.color_palette(\"deep\") # variations of default palette: deep, muted, pastel, bright, dark, colorblind. \nplt.style.use('seaborn-colorblind')\nsns.set(rc={'figure.figsize':(8,5)})\nsns.barplot(x=df_count_target.iloc[:,0], y=df_count_target.iloc[:,1],data=df_count_target, alpha = 0.6).set_title('Count of targets (Non-defaulters vs Defaulters)')\nplt.xlabel('Target')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.256922Z","iopub.execute_input":"2022-09-02T08:29:02.257254Z","iopub.status.idle":"2022-09-02T08:29:02.645354Z","shell.execute_reply.started":"2022-09-02T08:29:02.257224Z","shell.execute_reply":"2022-09-02T08:29:02.644269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Pipeline data by including feature engineering, train-test validation and machine learning models into an algorithm to facilitate model prediction.\n\nDue to having a large dataset, subset only 5% of the dataset instead of using big data tools (SQL, Hadoop, etc.) to reduce time taken to run different hyperparameter tunnings and machine learning algorithm to determine optimal prediction model that generates highest overall accuracy.\n\n***Note: Do remember to delete unnecessary or single-used data that was loaded in this kernel using 'del __ 'function and 'gc.collect()', to ensure that GPU, RAM and CPU does not exceed the limited capacity imposed by Kaggle that would fail to run codes and automatically restart kernel. This allows the latter codes that are in use and necessary to run and produce outcome of prediction model.***\n","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer #transform different types\n\n# Subsample 10% of dataset to reduce time at an expense of performance to ensure ML model runs efficiently fast.\ndf_train_sample = df_train.copy().sample(frac=0.1)\n# Pipe-line preprocessing for numerical and categorical features to prevent error in codes and leaking of data during training:\nfeatures = df_train_sample.drop('target', axis=1).columns\nX = df_train_sample[features].copy()\ny = df_train_sample['target'].copy()\n\nprint(df_train_sample)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.646795Z","iopub.execute_input":"2022-09-02T08:29:02.647141Z","iopub.status.idle":"2022-09-02T08:29:02.945185Z","shell.execute_reply.started":"2022-09-02T08:29:02.647108Z","shell.execute_reply":"2022-09-02T08:29:02.944121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_features = [ 'B_30','B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer()),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))])","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.946315Z","iopub.execute_input":"2022-09-02T08:29:02.946652Z","iopub.status.idle":"2022-09-02T08:29:02.954548Z","shell.execute_reply.started":"2022-09-02T08:29:02.946609Z","shell.execute_reply":"2022-09-02T08:29:02.951365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features = [x for x in features if x not in categorical_features]\nnumerical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer()),\n    ('scaler', StandardScaler())])\n","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.955986Z","iopub.execute_input":"2022-09-02T08:29:02.956367Z","iopub.status.idle":"2022-09-02T08:29:02.965397Z","shell.execute_reply.started":"2022-09-02T08:29:02.956324Z","shell.execute_reply":"2022-09-02T08:29:02.964329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_transformer = ColumnTransformer(\n    transformers=[\n        ('numerical', numerical_transformer, numerical_features),\n        ('categorical', categorical_transformer, categorical_features)], \n        remainder='passthrough') ","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.968983Z","iopub.execute_input":"2022-09-02T08:29:02.969321Z","iopub.status.idle":"2022-09-02T08:29:02.978449Z","shell.execute_reply.started":"2022-09-02T08:29:02.969291Z","shell.execute_reply":"2022-09-02T08:29:02.977243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, GridSearchCV   \n# Train-test validation approach.\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:02.979889Z","iopub.execute_input":"2022-09-02T08:29:02.980224Z","iopub.status.idle":"2022-09-02T08:29:03.028387Z","shell.execute_reply.started":"2022-09-02T08:29:02.980192Z","shell.execute_reply":"2022-09-02T08:29:03.027181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_train, df_train_sample, features\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:03.030452Z","iopub.execute_input":"2022-09-02T08:29:03.030980Z","iopub.status.idle":"2022-09-02T08:29:03.160976Z","shell.execute_reply.started":"2022-09-02T08:29:03.030933Z","shell.execute_reply":"2022-09-02T08:29:03.159443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set parameters for imputer on numerical and categorical features.\nparam_grid = {\n    'data_transformer__numerical__imputer__strategy': ['mean', 'median'] ,\n    'data_transformer__categorical__imputer__strategy': ['constant','most_frequent']\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:03.162733Z","iopub.execute_input":"2022-09-02T08:29:03.163120Z","iopub.status.idle":"2022-09-02T08:29:03.171554Z","shell.execute_reply.started":"2022-09-02T08:29:03.163083Z","shell.execute_reply":"2022-09-02T08:29:03.170145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# RandomForest Model\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error, accuracy_score\nfrom sklearn.ensemble import RandomForestClassifier\npipe_rfc = Pipeline(steps=[('data_transformer', data_transformer),\n                     ('pipe_rfc', RandomForestClassifier(random_state=0))])  # ,\n\ngrid_rfc = GridSearchCV(pipe_rfc, param_grid=param_grid, cv = 10) # , cv=10\ngrid_rfc.fit(X_train, y_train.values.ravel()); \n\nprint(grid_rfc.best_score_) \nprint(grid_rfc.best_params_)\nprint(grid_rfc.best_estimator_)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:29:03.173210Z","iopub.execute_input":"2022-09-02T08:29:03.174134Z","iopub.status.idle":"2022-09-02T08:48:40.167541Z","shell.execute_reply.started":"2022-09-02T08:29:03.174094Z","shell.execute_reply":"2022-09-02T08:48:40.166332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:48:40.169244Z","iopub.execute_input":"2022-09-02T08:48:40.169727Z","iopub.status.idle":"2022-09-02T08:48:40.309104Z","shell.execute_reply.started":"2022-09-02T08:48:40.169682Z","shell.execute_reply":"2022-09-02T08:48:40.308039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import ConfusionMatrixDisplay, confusion_matrix, classification_report\nimport matplotlib.pyplot as plt\n\ny_pred_rfc = grid_rfc.predict(X_test)\ny_pred_prob_rfc = grid_rfc.predict_proba(X_test)\ncm_rfc = confusion_matrix(y_test, y_pred_rfc, labels= grid_rfc.classes_)\n\ndisp_rfc = ConfusionMatrixDisplay(confusion_matrix=cm_rfc)\ndisp_rfc.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:48:40.310370Z","iopub.execute_input":"2022-09-02T08:48:40.310700Z","iopub.status.idle":"2022-09-02T08:48:41.349473Z","shell.execute_reply.started":"2022-09-02T08:48:40.310671Z","shell.execute_reply":"2022-09-02T08:48:41.348130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\n\nprint(classification_report(y_test, y_pred_rfc))\nprint('accuracy score:', accuracy_score(y_test, y_pred_rfc)) # overall accuracy score: true positive+true negative/total observations\nprint('recall score:', recall_score(y_test, y_pred_rfc, average=None)) # recall score for passenger that survives is not very good, recommend to adjust threshold to get better and balanced score.\nprint('precision score:', precision_score(y_test, y_pred_rfc, average=None))\nprint('f1 score:', f1_score(y_test, y_pred_rfc, average=None))","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:48:41.351673Z","iopub.execute_input":"2022-09-02T08:48:41.354962Z","iopub.status.idle":"2022-09-02T08:48:41.415841Z","shell.execute_reply.started":"2022-09-02T08:48:41.354917Z","shell.execute_reply":"2022-09-02T08:48:41.414638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It can be observed above that Random Forest classifier predicts better on customer who do not default (0) than customers who default (1). This is attirbuted to the imbalanced distribution of customer who do not and do default their credit cards.","metadata":{}},{"cell_type":"code","source":"precision, recall, threshold = precision_recall_curve(y_test.astype(int), grid_rfc.predict_proba(X_test)[:,1])\n\nplt.plot(threshold, precision[:-1], c='r', label='Precision')\nplt.plot(threshold, recall[:-1], c='b', label='Recall')\nplt.grid()\nplt.legend()\nplt.title('Precision-Recall Curve')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:48:41.417918Z","iopub.execute_input":"2022-09-02T08:48:41.419442Z","iopub.status.idle":"2022-09-02T08:48:42.080332Z","shell.execute_reply.started":"2022-09-02T08:48:41.419384Z","shell.execute_reply":"2022-09-02T08:48:42.079113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, plot_roc_curve\n\nax = plt.gca()  # get current Axes instance\nplot_roc_curve(grid_rfc, X_test, y_test, ax=ax, name='Random Forest Classifier')\nplt.show()\n\nprint('roc auc score:', roc_auc_score(y_test, grid_rfc.predict_proba(X_test)[:,1]))","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:48:42.081621Z","iopub.execute_input":"2022-09-02T08:48:42.082430Z","iopub.status.idle":"2022-09-02T08:48:43.144375Z","shell.execute_reply.started":"2022-09-02T08:48:42.082397Z","shell.execute_reply":"2022-09-02T08:48:43.143195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train, y_train , X_test, y_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:48:43.146129Z","iopub.execute_input":"2022-09-02T08:48:43.146474Z","iopub.status.idle":"2022-09-02T08:48:43.279024Z","shell.execute_reply.started":"2022-09-02T08:48:43.146442Z","shell.execute_reply":"2022-09-02T08:48:43.277855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre-process test dataset too\ndf_submission = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\", usecols=['customer_ID'], low_memory=True)\ndf_test = pd.read_parquet('../input/train-test-parquet/test.parquet')\nprint(len(df_submission))\nprint(len(df_test))","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:48:43.284173Z","iopub.execute_input":"2022-09-02T08:48:43.285078Z","iopub.status.idle":"2022-09-02T08:49:29.334917Z","shell.execute_reply.started":"2022-09-02T08:48:43.285040Z","shell.execute_reply":"2022-09-02T08:49:29.333855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.groupby('customer_ID').tail(1)\nprint(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:49:29.336291Z","iopub.execute_input":"2022-09-02T08:49:29.337132Z","iopub.status.idle":"2022-09-02T08:49:34.298464Z","shell.execute_reply.started":"2022-09-02T08:49:29.337095Z","shell.execute_reply":"2022-09-02T08:49:34.297257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.drop(df_test.iloc[:,0:2], axis=1, inplace=True)\ndf_test[['D_63','D_64']] = df_test[['D_63','D_64']].astype('category').apply(lambda x: x.cat.codes)\ndf_test = df_test.apply(abs)\n\n#df_test = df_test.drop(df_test.columns[df_test.apply(lambda col: ((col.isnull() | col.isna()).mean().round(5)*100) >= 50)], axis=1)\ncat_test_var = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ndf_test[cat_test_var] = df_test[cat_test_var].fillna(0)\n\nprint(df_test)\nprint(df_test.dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:49:34.300732Z","iopub.execute_input":"2022-09-02T08:49:34.301201Z","iopub.status.idle":"2022-09-02T08:49:35.511088Z","shell.execute_reply.started":"2022-09-02T08:49:34.301153Z","shell.execute_reply":"2022-09-02T08:49:35.509869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if columns are removed correctly based on number of columns with more than 50% of NAs\ncount_nulls = (df_test.isnull() | df_test.isna()).mean().round(5)*100\ncount_nulls = count_nulls.drop(count_nulls[count_nulls >= 50].index)\nprint(count_nulls)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:49:35.512416Z","iopub.execute_input":"2022-09-02T08:49:35.512806Z","iopub.status.idle":"2022-09-02T08:49:35.936980Z","shell.execute_reply.started":"2022-09-02T08:49:35.512772Z","shell.execute_reply":"2022-09-02T08:49:35.935637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.isnull().all(axis=1)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:49:35.938620Z","iopub.execute_input":"2022-09-02T08:49:35.938999Z","iopub.status.idle":"2022-09-02T08:49:36.221868Z","shell.execute_reply.started":"2022-09-02T08:49:35.938963Z","shell.execute_reply":"2022-09-02T08:49:36.220915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### As American Express wishes to predict customers who default their credit cards with the company, only account and keep the value of second column which is (1): customers who defaults.","metadata":{}},{"cell_type":"code","source":"# Predict on test data set\ntest_pred_rfc = grid_rfc.predict(df_test)\ntest_pred_prob_rfc = grid_rfc.predict_proba(df_test)\nprint(len(test_pred_prob_rfc))\nprint(test_pred_prob_rfc)\nprint(test_pred_prob_rfc[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:49:36.223324Z","iopub.execute_input":"2022-09-02T08:49:36.223665Z","iopub.status.idle":"2022-09-02T08:50:24.904646Z","shell.execute_reply.started":"2022-09-02T08:49:36.223635Z","shell.execute_reply":"2022-09-02T08:50:24.902969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['prediction'] = None\ndf_test['prediction'] = test_pred_prob_rfc[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:50:24.906264Z","iopub.execute_input":"2022-09-02T08:50:24.907180Z","iopub.status.idle":"2022-09-02T08:50:24.929972Z","shell.execute_reply.started":"2022-09-02T08:50:24.907123Z","shell.execute_reply":"2022-09-02T08:50:24.928840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'customer_ID': df_submission.customer_ID, 'prediction': test_pred_prob_rfc[:,1]})\nsubmission.to_csv('submission.csv', index=False)\nprint(submission)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:50:24.933831Z","iopub.execute_input":"2022-09-02T08:50:24.934885Z","iopub.status.idle":"2022-09-02T08:50:27.395479Z","shell.execute_reply.started":"2022-09-02T08:50:24.934841Z","shell.execute_reply":"2022-09-02T08:50:27.394245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Check if submission.csv file has the same format as sample_submission file.","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('./submission.csv')\nprint(submission)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T08:50:27.396827Z","iopub.execute_input":"2022-09-02T08:50:27.397195Z","iopub.status.idle":"2022-09-02T08:50:28.581131Z","shell.execute_reply.started":"2022-09-02T08:50:27.397160Z","shell.execute_reply":"2022-09-02T08:50:28.580225Z"},"trusted":true},"execution_count":null,"outputs":[]}]}