{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Import Packages","metadata":{}},{"cell_type":"code","source":"#load packages\nimport numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom sklearn import preprocessing, tree, model_selection, metrics\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.tree import DecisionTreeClassifier\n\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.metrics import confusion_matrix, classification_report, roc_auc_score, precision_recall_curve, roc_curve, auc, average_precision_score, ConfusionMatrixDisplay, precision_recall_fscore_support \n\n\n#install dataset package\n#!pip install --quiet dmba\n#import dmba\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nfrom sklearn.preprocessing import scale\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.decomposition import IncrementalPCA\n","metadata":{"execution":{"iopub.status.busy":"2022-11-25T10:58:01.017298Z","iopub.execute_input":"2022-11-25T10:58:01.017999Z","iopub.status.idle":"2022-11-25T10:58:01.031134Z","shell.execute_reply.started":"2022-11-25T10:58:01.017960Z","shell.execute_reply":"2022-11-25T10:58:01.029859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Steps\n\n**1. Read Data**  ✔ \n\n**2. Pre-Process and aggregate the data**  ✔ \n\n**3. Split variables into different groups**  ✔ \n\n**4. Perform Dimensionality Reduction in each group**  ✔ \n\n**5. Split data into train-test and validation**\n\n**6. Build Classifier Model and perform daignostics**\n\n**7. Hyper parameter tuning using Grid Search**\n\n**8. Model Validation**\n\n**9. Results**\n","metadata":{}},{"cell_type":"markdown","source":"### Load Data","metadata":{}},{"cell_type":"code","source":"### Loading Data\n#load data\namex_data = pd.read_parquet('/kaggle/input/amex-parquet/train_data.parquet')\nprint(\"The shape with the categorical features is:\", amex_data.shape)\n\n#get labels\namex_labels = amex_data[['customer_ID', 'target']] #pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv').iloc[:,1]z\namex_labels = amex_labels.groupby('customer_ID').agg({'target' : 'min'})\nprint(\"The shape of the data labels is:\", amex_labels.shape)\n\n\n#load test data\n#test_data = pd.read_parquet('/kaggle/input/amex-parquet/test_data.parquet')\n#print(\"The shape with the categorical features is:\", test_data.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:05:26.611662Z","iopub.execute_input":"2022-11-25T08:05:26.612277Z","iopub.status.idle":"2022-11-25T08:06:11.108384Z","shell.execute_reply.started":"2022-11-25T08:05:26.612240Z","shell.execute_reply":"2022-11-25T08:06:11.107392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:11.110698Z","iopub.execute_input":"2022-11-25T08:06:11.111333Z","iopub.status.idle":"2022-11-25T08:06:11.149324Z","shell.execute_reply.started":"2022-11-25T08:06:11.111292Z","shell.execute_reply":"2022-11-25T08:06:11.148157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(amex_data.customer_ID.unique())","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:11.153675Z","iopub.execute_input":"2022-11-25T08:06:11.155406Z","iopub.status.idle":"2022-11-25T08:06:11.958059Z","shell.execute_reply.started":"2022-11-25T08:06:11.155367Z","shell.execute_reply":"2022-11-25T08:06:11.957084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"></a>\n# <div style='display:fill;color:#dfe3e8;background-color:#053673;padding:20px'>   <b> DATA PRE-PROCESSING </b> </div>","metadata":{}},{"cell_type":"markdown","source":"### Remove categorical variables","metadata":{}},{"cell_type":"code","source":"CATEGORICAL_FEATURES = [\"B_30\", \"B_38\", \"D_114\", \"D_116\", \"D_117\", \"D_120\", \"D_126\", \"D_63\", \"D_64\", \"D_66\", \"D_68\"]\n\n#remove catagorical features\namex_data = amex_data.drop(CATEGORICAL_FEATURES, axis=1)\namex_data = amex_data.drop(['target'], axis=1)\nprint(\"The train data shape without the categorical features is:\", amex_data.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:11.963678Z","iopub.execute_input":"2022-11-25T08:06:11.966015Z","iopub.status.idle":"2022-11-25T08:06:21.087015Z","shell.execute_reply.started":"2022-11-25T08:06:11.965973Z","shell.execute_reply":"2022-11-25T08:06:21.085823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing columns with high Missing Data","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nmissing_data=pd.DataFrame((amex_data.isnull().sum()/len(amex_data))*100, columns=['% Missing'])\nfig, ax = plt.subplots(1,1, figsize=(20,6))\nsns.histplot(data=missing_data, x=\"% Missing\")\nax.set_ylabel('No. of Columns')\nax.tick_params(axis='x', rotation=90)\nplt.suptitle(\"Missing Data - frequency distribution [%]\", fontsize = 20)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:21.088638Z","iopub.execute_input":"2022-11-25T08:06:21.089033Z","iopub.status.idle":"2022-11-25T08:06:24.053289Z","shell.execute_reply.started":"2022-11-25T08:06:21.088997Z","shell.execute_reply":"2022-11-25T08:06:24.052312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_with_high_missing_data = list(missing_data[missing_data[\"% Missing\"] >80].index)\namex_data = amex_data.drop(cols_with_high_missing_data, axis=1)\nprint(\"Follwing columns have been removed from the dataset:\", cols_with_high_missing_data)\nprint(\"The amex data shape after removing features with high missing data is:\", amex_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:24.054854Z","iopub.execute_input":"2022-11-25T08:06:24.055489Z","iopub.status.idle":"2022-11-25T08:06:25.919186Z","shell.execute_reply.started":"2022-11-25T08:06:24.055445Z","shell.execute_reply":"2022-11-25T08:06:25.917899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Identifying other non-numeric columns (like date) and removing them\nWe notice that S_2 is a date varibale. We will aggregate the varibales to remove the time-component. Hence, we remoce the variable S_2","metadata":{}},{"cell_type":"code","source":"print('Non-numeric variables:', set(amex_data.columns) - set(amex_data._get_numeric_data().columns))\namex_data = amex_data.drop(['S_2'], axis=1)\nprint(\"The amex data shape after eliminating date variables:\", amex_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:25.920626Z","iopub.execute_input":"2022-11-25T08:06:25.920983Z","iopub.status.idle":"2022-11-25T08:06:27.972450Z","shell.execute_reply.started":"2022-11-25T08:06:25.920947Z","shell.execute_reply":"2022-11-25T08:06:27.971250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### DATA AGGREGATION","metadata":{}},{"cell_type":"code","source":"aggregation_dict = dict.fromkeys(list(set(amex_data.columns) - set(['customer_ID'])), \"mean\")\nagg_amex_data = amex_data.groupby(['customer_ID']).agg(aggregation_dict)\nagg_amex_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:27.976647Z","iopub.execute_input":"2022-11-25T08:06:27.979295Z","iopub.status.idle":"2022-11-25T08:06:43.085135Z","shell.execute_reply.started":"2022-11-25T08:06:27.979251Z","shell.execute_reply":"2022-11-25T08:06:43.084140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing value treatment - Filling missing data with corresponding column Median\nfor column in agg_amex_data.columns:\n    median = agg_amex_data[column].median()\n    agg_amex_data[column] = agg_amex_data[column].fillna(median)\n    \n    \nagg_amex_data.head(5)    ","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:43.089765Z","iopub.execute_input":"2022-11-25T08:06:43.092285Z","iopub.status.idle":"2022-11-25T08:06:44.440623Z","shell.execute_reply.started":"2022-11-25T08:06:43.092243Z","shell.execute_reply":"2022-11-25T08:06:44.439448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"></a>\n# <div style='display:fill;color:#dfe3e8;background-color:#053673;padding:20px'>   <b> Feature Engineering & Feature Selection</b> </div>","metadata":{}},{"cell_type":"markdown","source":"### DIMENSIONALITY REDUCTION","metadata":{}},{"cell_type":"code","source":"## Grouping columns by categories\n\nDeliquency_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='D']\nSpend_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='S']\nPayment_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='P']\nBalance_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='B']\nRisk_variables = [x for x in set(agg_amex_data.columns) - {'customer_ID'} if x[0]=='R']","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:44.445288Z","iopub.execute_input":"2022-11-25T08:06:44.445630Z","iopub.status.idle":"2022-11-25T08:06:44.452580Z","shell.execute_reply.started":"2022-11-25T08:06:44.445601Z","shell.execute_reply":"2022-11-25T08:06:44.451456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Deliquency**","metadata":{}},{"cell_type":"code","source":"## PCA on Deliqunecy\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Deliquency_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \n#x = sc.fit_transform(x)\n#x_test = sc.transform(x_test)\n\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=70)\npcs.fit(x)\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(range(1, 71), [0.9] * 70, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, 71)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:44.454247Z","iopub.execute_input":"2022-11-25T08:06:44.454626Z","iopub.status.idle":"2022-11-25T08:06:54.020446Z","shell.execute_reply.started":"2022-11-25T08:06:44.454564Z","shell.execute_reply":"2022-11-25T08:06:54.019559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Deliquency_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\ndeliquency_variables = pca_features[['Deliquency_{}'.format(i) for i in range(1, 39 )]]\ndeliquency_variables","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:54.024606Z","iopub.execute_input":"2022-11-25T08:06:54.026695Z","iopub.status.idle":"2022-11-25T08:06:56.242363Z","shell.execute_reply.started":"2022-11-25T08:06:54.026658Z","shell.execute_reply":"2022-11-25T08:06:56.241234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Spend**","metadata":{}},{"cell_type":"code","source":"## PCA on Spend\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Spend_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \n#x = sc.fit_transform(x)\n#x_test = sc.transform(x_test)\n\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=20)\npcs.fit(x)\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(list(range(0,20 )), [0.9] * 20, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\n\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, 21)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:56.243804Z","iopub.execute_input":"2022-11-25T08:06:56.244635Z","iopub.status.idle":"2022-11-25T08:06:57.957688Z","shell.execute_reply.started":"2022-11-25T08:06:56.244595Z","shell.execute_reply":"2022-11-25T08:06:57.956630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Spend_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\nspend_variables = pca_features[['Spend_{}'.format(i) for i in range(1, 9 )]]\nspend_variables","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:57.959374Z","iopub.execute_input":"2022-11-25T08:06:57.959764Z","iopub.status.idle":"2022-11-25T08:06:58.276223Z","shell.execute_reply.started":"2022-11-25T08:06:57.959727Z","shell.execute_reply":"2022-11-25T08:06:58.275036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Payment**","metadata":{}},{"cell_type":"code","source":"## PCA on Payment\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Payment_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \n#x = sc.fit_transform(x)\n#x_test = sc.transform(x_test)\n\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=3)\npcs.fit(x)\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(list(range(0,3 )), [0.9] * 3, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\n\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, 4)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:58.278250Z","iopub.execute_input":"2022-11-25T08:06:58.279018Z","iopub.status.idle":"2022-11-25T08:06:59.013639Z","shell.execute_reply.started":"2022-11-25T08:06:58.278975Z","shell.execute_reply":"2022-11-25T08:06:59.012748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Payment_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\npayment_variables = pca_features[['Payment_{}'.format(i) for i in range(1, 4 )]]\npayment_variables","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:59.015207Z","iopub.execute_input":"2022-11-25T08:06:59.015592Z","iopub.status.idle":"2022-11-25T08:06:59.068459Z","shell.execute_reply.started":"2022-11-25T08:06:59.015555Z","shell.execute_reply":"2022-11-25T08:06:59.066986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Balance**","metadata":{}},{"cell_type":"code","source":"## PCA on Balance\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Balance_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \n#x = sc.fit_transform(x)\n#x_test = sc.transform(x_test)\n\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=20)\npcs.fit(x)\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(list(range(0,20 )), [0.9] * 20, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\n\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:06:59.074150Z","iopub.execute_input":"2022-11-25T08:06:59.075148Z","iopub.status.idle":"2022-11-25T08:07:06.254657Z","shell.execute_reply.started":"2022-11-25T08:06:59.075086Z","shell.execute_reply":"2022-11-25T08:07:06.253610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Balance_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\nbalance_variables = pca_features[['Balance_{}'.format(i) for i in range(1, 4 )]]\nbalance_variables","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:07:06.256485Z","iopub.execute_input":"2022-11-25T08:07:06.256859Z","iopub.status.idle":"2022-11-25T08:07:10.489138Z","shell.execute_reply.started":"2022-11-25T08:07:06.256822Z","shell.execute_reply":"2022-11-25T08:07:10.488099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Risk**","metadata":{}},{"cell_type":"code","source":"## PCA on Risk\n# distributing the dataset into two components X and Y\nx = agg_amex_data[Risk_variables].values\nplt.figure(figsize = (18,10))        \nsns.heatmap(pd.DataFrame(x).corr(),annot = False,cmap=\"YlGnBu\")\n\n\n# Standardizing the Variables\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\n  \n#x = sc.fit_transform(x)\n#x_test = sc.transform(x_test)\n\n\n# Prinicpal Component Analysis\npcs = PCA(n_components=20)\npcs.fit(x)\n\npcsSummary = pd.DataFrame({'Standard deviation': np.sqrt(pcs.explained_variance_),\n'Proportion of variance': pcs.explained_variance_ratio_,\n'Cumulative proportion': np.cumsum(pcs.explained_variance_ratio_)})\n\n## Cumulative Variance chart\nFigure = plt.figure(figsize = (20,6))\nplt.plot(np.cumsum(pcs.explained_variance_ratio_))\nplt.plot(list(range(0,20 )), [0.9] * 20, label = \"threshold\")\nplt.xlabel('Components')\nplt.ylabel('Cumulative Variance')\n\n\n\npcsSummary = pcsSummary.transpose()\npcsSummary.columns = ['PC{}'.format(i) for i in range(1, 21)]\npcsSummary.round(4)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:07:10.490814Z","iopub.execute_input":"2022-11-25T08:07:10.491236Z","iopub.status.idle":"2022-11-25T08:07:14.346183Z","shell.execute_reply.started":"2022-11-25T08:07:10.491197Z","shell.execute_reply":"2022-11-25T08:07:14.345138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_features = pd.DataFrame(pcs.fit_transform(x))\npca_features.columns = ['Risk_{}'.format(i) for i in range(1, len(pcsSummary.columns) + 1)]\npca_features.set_index(agg_amex_data.index,inplace=True)\nrisk_variables = pca_features[['Risk_{}'.format(i) for i in range(1, 15 )]]\nrisk_variables","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:07:14.349183Z","iopub.execute_input":"2022-11-25T08:07:14.349472Z","iopub.status.idle":"2022-11-25T08:07:17.024795Z","shell.execute_reply.started":"2022-11-25T08:07:14.349444Z","shell.execute_reply":"2022-11-25T08:07:17.023822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Final Data after dimensionality reduction","metadata":{}},{"cell_type":"code","source":"amex_data_red = pd.concat([deliquency_variables, spend_variables, payment_variables, balance_variables, risk_variables], axis = 1)\namex_data_processed = amex_data_red.join(amex_labels)\namex_data_processed","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:07:17.029801Z","iopub.execute_input":"2022-11-25T08:07:17.032485Z","iopub.status.idle":"2022-11-25T08:07:18.008265Z","shell.execute_reply.started":"2022-11-25T08:07:17.032432Z","shell.execute_reply":"2022-11-25T08:07:18.007224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#amex_data_processed.to_csv('processed_data.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:07:53.362642Z","iopub.execute_input":"2022-11-25T08:07:53.363043Z","iopub.status.idle":"2022-11-25T08:08:19.303986Z","shell.execute_reply.started":"2022-11-25T08:07:53.363011Z","shell.execute_reply":"2022-11-25T08:08:19.302809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"></a>\n# <div style='display:fill;color:#dfe3e8;background-color:#053673;padding:20px'>   <b> Feature Engineering & Selection</b> </div>","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"<a id=\"5\"></a>\n# <div style='display:fill;color:#dfe3e8;background-color:#053673;padding:20px'>   <b> Model Building </b> </div>","metadata":{}},{"cell_type":"markdown","source":"# Split Data into Train & Test","metadata":{}},{"cell_type":"code","source":"x = amex_data_processed.loc[:, amex_data_processed.columns != 'target']\ny = amex_data_processed['target']\nx_train, x_test, y_train, y_test = train_test_split(x, y,stratify=y,test_size=0.20, random_state=10)\n\nprint('Train Data size: ',len(x_train))\nprint('Test Data size: ',len(x_test))","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:15:01.680119Z","iopub.execute_input":"2022-11-25T08:15:01.680853Z","iopub.status.idle":"2022-11-25T08:15:02.108286Z","shell.execute_reply.started":"2022-11-25T08:15:01.680816Z","shell.execute_reply":"2022-11-25T08:15:02.107251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.countplot(data = y_train)\ntemp1 = pd.DataFrame(y_train,columns=['target'])\ntemp2 = pd.DataFrame(y_test,columns=['target'])\n\n\n## Plot the charts\nfig,plts = plt.subplots(1,2, figsize=(20,6))\n#fig, (plt1,plt2,plt3) = plt.subplots(1, 3, figsize=(20, 6))\nfig.suptitle('Distribution of target variable', size=15)\nsns.countplot(x='target',data=temp1, ax=plts[0])\nplts[0].set(xlabel='Target', ylabel='No. of Customers', title='Train Data')\nabs_values = temp1['target'].value_counts(ascending=False)\nrel_values = temp1['target'].value_counts(ascending=False, normalize=True).values * 100\nlbls = [f'{p[0]} ({p[1]:.0f}%)' for p in zip(abs_values, rel_values)]\nplts[0].bar_label(container=plts[0].containers[0], labels=lbls)\n\n\n\n\n\nsns.countplot(x='target',data=temp2, ax=plts[1])\nplts[1].set(xlabel='Target', ylabel='No. of Customers', title='Test Data')\nabs_values = temp2['target'].value_counts(ascending=False)\nrel_values = temp2['target'].value_counts(ascending=False, normalize=True).values * 100\nlbls = [f'{p[0]} ({p[1]:.0f}%)' for p in zip(abs_values, rel_values)]\nplts[1].bar_label(container=plts[1].containers[0], labels=lbls)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:37:58.087657Z","iopub.execute_input":"2022-11-25T08:37:58.088211Z","iopub.status.idle":"2022-11-25T08:37:58.880401Z","shell.execute_reply.started":"2022-11-25T08:37:58.088155Z","shell.execute_reply":"2022-11-25T08:37:58.879458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decision Tree","metadata":{}},{"cell_type":"code","source":"## Model initialization\nmtype = 'Decision Tree'\ncutoffs = [0.3,0.4,0.5,0.6,0.7]\ncutoff = []\ntrain_acc = []\ntest_acc= []\nroc_auc = []\nprecision = []\nrecall= []\nf1 = []\n\n## Define Classifier\nclf = DecisionTreeClassifier(random_state=10)\nmodel = clf.fit(x_train, y_train)\n\n\n## Make Predictions\ny_pred_prob = model.predict_proba(x_test)[:,1]\ny_pred_prob_train = model.predict_proba(x_train)[:,1]\nfig1, ax = plt.subplots(1, 5, figsize=(20, 3),constrained_layout = True)\n\n\n## Performance of the model\nfor i in range(0, len(cutoffs)):\n    y_pred = (y_pred_prob>=cutoffs[i])*1\n    y_pred_train = (y_pred_prob_train>=cutoffs[i])*1\n\n    #  Performance Report\n    cutoff.append(cutoffs[i])\n    train_acc.append(round(sum((y_pred_train==y_train)*1)/len(y_pred_train),4))\n    test_acc.append(round(sum((y_pred==y_test)*1)/len(y_pred),4))\n    roc_auc.append(round(roc_auc_score(y_test, y_pred),4))\n     \n    \n    #  Confusion Matrix\n    cm = confusion_matrix(y_test, y_pred, labels=clf.classes_)\n    ConfusionMatrixDisplay(confusion_matrix=cm,display_labels=clf.classes_ ).plot(ax=ax[i],cmap=\"YlGnBu\")\n    ax[i].set(title='Confusion matrix at cutoff: '+ str(cutoffs[i]))\n      \n   #   Classification Report\n    prfs = precision_recall_fscore_support(y_test, y_pred, average='binary',pos_label=1)\n    precision.append(round(prfs[0],4))\n    recall.append(round(prfs[1],4))\n    f1.append(round(prfs[2],4))\n\nperf = pd.DataFrame(list(zip(cutoff, train_acc, test_acc, roc_auc)), columns =['Prob Cutoff', 'Training Accuracy', 'Test Accuracy', 'ROC AUC Score'])\nprfs = pd.DataFrame(list(zip(cutoff, precision, recall, f1)), columns =['Prob Cutoff', 'Precision', 'Recall', 'F1 Score'])\n\n\nprint(\"-\"*15,\"PERFORMANCE REPORT\",\"-\"*15)\nprint(perf)\nprint(\"-\"*15,\"CLASSIFICATION REPORT\",\"-\"*15)\nprint(prfs)\n\n# precision-recall curve\nfig2, (ax1, ax2) = plt.subplots(1,2,figsize = (12,6),constrained_layout = True)\n    \nprecision, recall, thresholds_pr = precision_recall_curve(y_test, y_pred_prob)\navg_pre = average_precision_score(y_test, y_pred_prob)\nax1.plot(precision, recall, label = mtype+ \" average precision = {:0.2f}\".format(avg_pre), lw = 3, alpha = 0.7)\nax1.set_xlabel('Precision', fontsize = 14)\nax1.set_ylabel('Recall', fontsize = 14)\nax1.set_title('Precision-Recall Curve', fontsize = 18)\nax1.legend(loc = 'best')\n#find default threshold\nclose_default = np.argmin(np.abs(thresholds_pr - 0.5))\nax1.plot(precision[close_default], recall[close_default], 'o', markersize = 8)\n\n# roc-curve\nfpr, tpr, thresholds_roc = roc_curve(y_test, y_pred_prob)\nroc_auc = auc(fpr,tpr)\nax2.plot(fpr,tpr, label = mtype+ \" area = {:0.2f}\".format(roc_auc), lw = 3, alpha = 0.7)\nax2.plot([0,1], [0,1], 'r', linestyle = \"--\", lw = 2)\nax2.set_xlabel(\"False Positive Rate\", fontsize = 14)\nax2.set_ylabel(\"True Positive Rate\", fontsize = 14)\nax2.set_title(\"ROC Curve\", fontsize = 18)\nax2.legend(loc = 'best')\n\n\n# find default threshold\nclose_default = np.argmin(np.abs(thresholds_roc - 0.5))\nax2.plot(fpr[close_default], tpr[close_default], 'o', markersize = 8)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-11-25T09:29:12.323558Z","iopub.execute_input":"2022-11-25T09:29:12.323907Z","iopub.status.idle":"2022-11-25T09:29:13.570997Z","shell.execute_reply.started":"2022-11-25T09:29:12.323875Z","shell.execute_reply":"2022-11-25T09:29:13.570077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest","metadata":{}},{"cell_type":"code","source":"## Model initialization\nmtype = 'Random Forest'\ncutoffs = [0.3,0.4,0.5,0.6,0.7]\ncutoff = []\ntrain_acc = []\ntest_acc= []\nroc_auc = []\nprecision = []\nrecall= []\nf1 = []\n\n## Define Classifier\nclf = RandomForestClassifier(random_state=10)\nmodel = clf.fit(x_train, y_train)\n\n\n## Make Predictions\ny_pred_prob = model.predict_proba(x_test)[:,1]\ny_pred_prob_train = model.predict_proba(x_train)[:,1]\nfig1, ax = plt.subplots(1, 5, figsize=(20, 3),constrained_layout = True)\n\n\n## Performance of the model\nfor i in range(0, len(cutoffs)):\n    y_pred = (y_pred_prob>=cutoffs[i])*1\n    y_pred_train = (y_pred_prob_train>=cutoffs[i])*1\n\n    #  Performance Report\n    cutoff.append(cutoffs[i])\n    train_acc.append(round(sum((y_pred_train==y_train)*1)/len(y_pred_train),4))\n    test_acc.append(round(sum((y_pred==y_test)*1)/len(y_pred),4))\n    roc_auc.append(round(roc_auc_score(y_test, y_pred),4))\n     \n    \n    #  Confusion Matrix\n    cm = confusion_matrix(y_test, y_pred, labels=clf.classes_)\n    ConfusionMatrixDisplay(confusion_matrix=cm,display_labels=clf.classes_ ).plot(ax=ax[i],cmap=\"YlGnBu\")\n    ax[i].set(title='Confusion matrix at cutoff: '+ str(cutoffs[i]))\n      \n   #   Classification Report\n    prfs = precision_recall_fscore_support(y_test, y_pred, average='binary',pos_label=1)\n    precision.append(round(prfs[0],4))\n    recall.append(round(prfs[1],4))\n    f1.append(round(prfs[2],4))\n\nperf = pd.DataFrame(list(zip(cutoff, train_acc, test_acc, roc_auc)), columns =['Prob Cutoff', 'Training Accuracy', 'Test Accuracy', 'ROC AUC Score'])\nprfs = pd.DataFrame(list(zip(cutoff, precision, recall, f1)), columns =['Prob Cutoff', 'Precision', 'Recall', 'F1 Score'])\n\n\nprint(\"-\"*15,\"PERFORMANCE REPORT\",\"-\"*15)\nprint(perf)\nprint(\"-\"*15,\"CLASSIFICATION REPORT\",\"-\"*15)\nprint(prfs)\n\n# precision-recall curve\nfig2, (ax1, ax2) = plt.subplots(1,2,figsize = (12,6),constrained_layout = True)\n    \nprecision, recall, thresholds_pr = precision_recall_curve(y_test, y_pred_prob)\navg_pre = average_precision_score(y_test, y_pred_prob)\nax1.plot(precision, recall, label = mtype+ \" average precision = {:0.2f}\".format(avg_pre), lw = 3, alpha = 0.7)\nax1.set_xlabel('Precision', fontsize = 14)\nax1.set_ylabel('Recall', fontsize = 14)\nax1.set_title('Precision-Recall Curve', fontsize = 18)\nax1.legend(loc = 'best')\n#find default threshold\nclose_default = np.argmin(np.abs(thresholds_pr - 0.5))\nax1.plot(precision[close_default], recall[close_default], 'o', markersize = 8)\n\n# roc-curve\nfpr, tpr, thresholds_roc = roc_curve(y_test, y_pred_prob)\nroc_auc = auc(fpr,tpr)\nax2.plot(fpr,tpr, label = mtype+ \" area = {:0.2f}\".format(roc_auc), lw = 3, alpha = 0.7)\nax2.plot([0,1], [0,1], 'r', linestyle = \"--\", lw = 2)\nax2.set_xlabel(\"False Positive Rate\", fontsize = 14)\nax2.set_ylabel(\"True Positive Rate\", fontsize = 14)\nax2.set_title(\"ROC Curve\", fontsize = 18)\nax2.legend(loc = 'best')\n\n\n# find default threshold\nclose_default = np.argmin(np.abs(thresholds_roc - 0.5))\nax2.plot(fpr[close_default], tpr[close_default], 'o', markersize = 8)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-11-25T08:49:54.621386Z","iopub.execute_input":"2022-11-25T08:49:54.622521Z","iopub.status.idle":"2022-11-25T09:00:39.524920Z","shell.execute_reply.started":"2022-11-25T08:49:54.622460Z","shell.execute_reply":"2022-11-25T09:00:39.524032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gradient Boosting","metadata":{}},{"cell_type":"code","source":"## Model initialization\nmtype = 'Gradient Boosting'\ncutoffs = [0.3,0.4,0.5,0.6,0.7]\ncutoff = []\ntrain_acc = []\ntest_acc= []\nroc_auc = []\nprecision = []\nrecall= []\nf1 = []\n\n## Define Classifier\nclf = GradientBoostingClassifier(random_state=10)\nmodel = clf.fit(x_train, y_train)\n\n\n## Make Predictions\ny_pred_prob = model.predict_proba(x_test)[:,1]\ny_pred_prob_train = model.predict_proba(x_train)[:,1]\nfig1, ax = plt.subplots(1, 5, figsize=(20, 3),constrained_layout = True)\n\n\n## Performance of the model\nfor i in range(0, len(cutoffs)):\n    y_pred = (y_pred_prob>=cutoffs[i])*1\n    y_pred_train = (y_pred_prob_train>=cutoffs[i])*1\n\n    #  Performance Report\n    cutoff.append(cutoffs[i])\n    train_acc.append(round(sum((y_pred_train==y_train)*1)/len(y_pred_train),4))\n    test_acc.append(round(sum((y_pred==y_test)*1)/len(y_pred),4))\n    roc_auc.append(round(roc_auc_score(y_test, y_pred),4))\n     \n    \n    #  Confusion Matrix\n    cm = confusion_matrix(y_test, y_pred, labels=clf.classes_)\n    ConfusionMatrixDisplay(confusion_matrix=cm,display_labels=clf.classes_ ).plot(ax=ax[i],cmap=\"YlGnBu\")\n    ax[i].set(title='Confusion matrix at cutoff: '+ str(cutoffs[i]))\n      \n   #   Classification Report\n    prfs = precision_recall_fscore_support(y_test, y_pred, average='binary',pos_label=1)\n    precision.append(round(prfs[0],4))\n    recall.append(round(prfs[1],4))\n    f1.append(round(prfs[2],4))\n\nperf = pd.DataFrame(list(zip(cutoff, train_acc, test_acc, roc_auc)), columns =['Prob Cutoff', 'Training Accuracy', 'Test Accuracy', 'ROC AUC Score'])\nprfs = pd.DataFrame(list(zip(cutoff, precision, recall, f1)), columns =['Prob Cutoff', 'Precision', 'Recall', 'F1 Score'])\n\n\nprint(\"-\"*15,\"PERFORMANCE REPORT\",\"-\"*15)\nprint(perf)\nprint(\"-\"*15,\"CLASSIFICATION REPORT\",\"-\"*15)\nprint(prfs)\n\n# precision-recall curve\nfig2, (ax1, ax2) = plt.subplots(1,2,figsize = (12,6),constrained_layout = True)\n    \nprecision, recall, thresholds_pr = precision_recall_curve(y_test, y_pred_prob)\navg_pre = average_precision_score(y_test, y_pred_prob)\nax1.plot(precision, recall, label = mtype+ \" average precision = {:0.2f}\".format(avg_pre), lw = 3, alpha = 0.7)\nax1.set_xlabel('Precision', fontsize = 14)\nax1.set_ylabel('Recall', fontsize = 14)\nax1.set_title('Precision-Recall Curve', fontsize = 18)\nax1.legend(loc = 'best')\n#find default threshold\nclose_default = np.argmin(np.abs(thresholds_pr - 0.5))\nax1.plot(precision[close_default], recall[close_default], 'o', markersize = 8)\n\n# roc-curve\nfpr, tpr, thresholds_roc = roc_curve(y_test, y_pred_prob)\nroc_auc = auc(fpr,tpr)\nax2.plot(fpr,tpr, label = mtype+ \" area = {:0.2f}\".format(roc_auc), lw = 3, alpha = 0.7)\nax2.plot([0,1], [0,1], 'r', linestyle = \"--\", lw = 2)\nax2.set_xlabel(\"False Positive Rate\", fontsize = 14)\nax2.set_ylabel(\"True Positive Rate\", fontsize = 14)\nax2.set_title(\"ROC Curve\", fontsize = 18)\nax2.legend(loc = 'best')\n\n\n# find default threshold\nclose_default = np.argmin(np.abs(thresholds_roc - 0.5))\nax2.plot(fpr[close_default], tpr[close_default], 'o', markersize = 8)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-11-25T11:43:00.382345Z","iopub.execute_input":"2022-11-25T11:43:00.382762Z","iopub.status.idle":"2022-11-25T11:43:03.939986Z","shell.execute_reply.started":"2022-11-25T11:43:00.382731Z","shell.execute_reply":"2022-11-25T11:43:03.939065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thresholds_roc - 0.3","metadata":{"execution":{"iopub.status.busy":"2022-11-25T11:33:11.143149Z","iopub.execute_input":"2022-11-25T11:33:11.144129Z","iopub.status.idle":"2022-11-25T11:33:11.151085Z","shell.execute_reply.started":"2022-11-25T11:33:11.144091Z","shell.execute_reply":"2022-11-25T11:33:11.150006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(y_pred_prob>=0.5)*1","metadata":{"execution":{"iopub.status.busy":"2022-11-25T10:27:41.953530Z","iopub.execute_input":"2022-11-25T10:27:41.953901Z","iopub.status.idle":"2022-11-25T10:27:41.962201Z","shell.execute_reply.started":"2022-11-25T10:27:41.953869Z","shell.execute_reply":"2022-11-25T10:27:41.961086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_prob>=0.5","metadata":{"execution":{"iopub.status.busy":"2022-11-25T10:27:45.514895Z","iopub.execute_input":"2022-11-25T10:27:45.515281Z","iopub.status.idle":"2022-11-25T10:27:45.522775Z","shell.execute_reply.started":"2022-11-25T10:27:45.515250Z","shell.execute_reply":"2022-11-25T10:27:45.521770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum((y_pred==y_test)*1)/len(y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-25T10:32:40.207431Z","iopub.execute_input":"2022-11-25T10:32:40.207792Z","iopub.status.idle":"2022-11-25T10:32:40.222668Z","shell.execute_reply.started":"2022-11-25T10:32:40.207762Z","shell.execute_reply":"2022-11-25T10:32:40.221290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}