{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3727003,"sourceType":"datasetVersion","datasetId":2213609}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-15T14:14:58.145555Z","iopub.execute_input":"2024-04-15T14:14:58.146003Z","iopub.status.idle":"2024-04-15T14:14:59.328611Z","shell.execute_reply.started":"2024-04-15T14:14:58.145964Z","shell.execute_reply":"2024-04-15T14:14:59.325514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyforest\n\nfrom pyforest import *","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:14:59.331184Z","iopub.execute_input":"2024-04-15T14:14:59.331837Z","iopub.status.idle":"2024-04-15T14:15:18.502059Z","shell.execute_reply.started":"2024-04-15T14:14:59.331794Z","shell.execute_reply":"2024-04-15T14:15:18.500971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Data","metadata":{}},{"cell_type":"markdown","source":"The competition's dataset is notably large, causing issues with memory capacity when reading the original csv files. Consequently, I use the AMEX-Feather-Dataset from @munumbutt, which compresses the data.I found this datasets whiles combing through the competition looking for ways to minimize the size of the data. This Feather file converts the floating-point precision from 64 bit to 16 bit, enhancing memory efficiency. Additionally, the Feather file format, being binary, allows for quicker data loading compared to csv files.\n","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_feather('../input/amexfeather/train_data.ftr')\ntest_data = pd.read_feather('../input/amexfeather/test_data.ftr')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:15:18.508561Z","iopub.execute_input":"2024-04-15T14:15:18.508930Z","iopub.status.idle":"2024-04-15T14:16:28.308073Z","shell.execute_reply.started":"2024-04-15T14:15:18.508893Z","shell.execute_reply":"2024-04-15T14:16:28.306628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:28.309843Z","iopub.execute_input":"2024-04-15T14:16:28.310369Z","iopub.status.idle":"2024-04-15T14:16:28.320828Z","shell.execute_reply.started":"2024-04-15T14:16:28.310335Z","shell.execute_reply":"2024-04-15T14:16:28.318644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see the data is very large ","metadata":{}},{"cell_type":"markdown","source":"## Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:28.322187Z","iopub.execute_input":"2024-04-15T14:16:28.322842Z","iopub.status.idle":"2024-04-15T14:16:30.119639Z","shell.execute_reply.started":"2024-04-15T14:16:28.322802Z","shell.execute_reply":"2024-04-15T14:16:30.116350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nprint(f\"There are {train_data.shape[1]-1} features in the data. {len(cat_features)} of them are categorical features.\")\n\nprint(f\"There are {train_data.shape[0]/1000000} million rows in the training data and {test_data.shape[0]/1000000} million rows in the test data.\")\nprint(f\"There are {train_data.customer_ID.nunique()} unique customers in the training data.\")\n\nprint(f\"There are {train_labels.shape[0]} customer_IDs in the train_labels.\")\nprint(f\"There are {train_labels.isna().sum().sum()} missing data and {train_labels.customer_ID.duplicated().sum()} duplicated customer_IDs in the train_labels.\")","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:30.125027Z","iopub.execute_input":"2024-04-15T14:16:30.125587Z","iopub.status.idle":"2024-04-15T14:16:31.386171Z","shell.execute_reply.started":"2024-04-15T14:16:30.125538Z","shell.execute_reply":"2024-04-15T14:16:31.384700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" #Display basic information\nprint(train_data.info())","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:31.387683Z","iopub.execute_input":"2024-04-15T14:16:31.388041Z","iopub.status.idle":"2024-04-15T14:16:31.425230Z","shell.execute_reply.started":"2024-04-15T14:16:31.387994Z","shell.execute_reply":"2024-04-15T14:16:31.423734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dealing with missing Data\n\n\nThe dataset contains a substantial amount of missing data. It is impractical to eliminate any column or row containing missing values entirely.\n\nExamining the 30 columns that have the highest number of null values.","metadata":{}},{"cell_type":"code","source":"null= pd.DataFrame(train_data.isnull().sum(),columns=['number_of_nulls'])\nnull['percentage_of_null'] = round(((null['number_of_nulls']/len(train_data))*100) , 2)\nnull = null[null['number_of_nulls']>0]\nnull= null.sort_values(by='percentage_of_null',ascending=False)\nnull.head(30)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:31.427134Z","iopub.execute_input":"2024-04-15T14:16:31.427584Z","iopub.status.idle":"2024-04-15T14:16:37.025190Z","shell.execute_reply.started":"2024-04-15T14:16:31.427549Z","shell.execute_reply":"2024-04-15T14:16:37.023628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the top null columns are delinquency features.","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,30))\nsns.barplot(data=null,y=null.index,x=null['percentage_of_null'])\ndel null","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:37.032431Z","iopub.execute_input":"2024-04-15T14:16:37.033740Z","iopub.status.idle":"2024-04-15T14:16:40.354692Z","shell.execute_reply.started":"2024-04-15T14:16:37.033674Z","shell.execute_reply":"2024-04-15T14:16:40.350637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Insights\n\nThere are many missing values, especially in the delinquency variables, with D_87 having the highest number of nulls (99.93%).\n","metadata":{}},{"cell_type":"markdown","source":"## Dealing on the Categorical Values","metadata":{}},{"cell_type":"code","source":"# Define a list of column names that are considered categorical variables\ncategorical_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\n# Initialize an empty list to hold the names of numerical columns\nnum_cols = []\n\n# Loop through each column in the dataframe 'df_train'\nfor col in train_data.columns:\n    # If the column is not in the list of categorical columns and is not 'customer_ID' or 'S_2',\n    # it is treated as a numerical column\n    if col not in categorical_cols + ['customer_ID', 'S_2']:\n        # Add the numerical column to the list\n        num_cols.append(col)\n\n# Convert the list of numerical columns to a numpy array for efficiency and better handling\nnum_cols = np.array(num_cols)\n\n# Print out the lists of categorical and numerical columns for verification\nprint(\"categorical cols: \", categorical_cols)\nprint(\"numerical cols: \", num_cols)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:40.356748Z","iopub.execute_input":"2024-04-15T14:16:40.357210Z","iopub.status.idle":"2024-04-15T14:16:40.369957Z","shell.execute_reply.started":"2024-04-15T14:16:40.357175Z","shell.execute_reply":"2024-04-15T14:16:40.368557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[categorical_cols]","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:40.371780Z","iopub.execute_input":"2024-04-15T14:16:40.372297Z","iopub.status.idle":"2024-04-15T14:16:40.472752Z","shell.execute_reply.started":"2024-04-15T14:16:40.372251Z","shell.execute_reply":"2024-04-15T14:16:40.471349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nIn the generated plot, I illustrate the distribution of features across various financial categories using a bar chart. Each category, such as Delinquency, Spend, Payment, Balance, and Risk, is represented along the x-axis, while the count of columns belonging to each category is depicted along the y-axis.\n\nThe height of each colored bar corresponds to the number of columns within that particular financial category. This visualization aids in understanding how the features are distributed across different aspects of financial data within the dataset. Despite including a legend in the code, it may not provide additional insights since there are no specific labels associated with each category.","metadata":{}},{"cell_type":"code","source":"# Set the figure size for the plot\nplt.figure(figsize=(16, 16))\n\n# Iterate through each categorical feature\nfor i, f in enumerate(cat_features):\n    # Create subplots with 4 rows, 3 columns, and increment the subplot index by 1\n    plt.subplot(4, 3, i+1)\n    \n    # Calculate and plot the distribution of values for target=0\n    temp = pd.DataFrame(train_data[f][train_data.target == 0].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp.index.name = 'value'\n    temp.reset_index(inplace=True)\n    plt.bar(temp.index, temp['count'], alpha=0.5, label='target=0')\n    \n    # Calculate and plot the distribution of values for target=1\n    temp = pd.DataFrame(train_data[f][train_data.target == 1].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n    temp.index.name = 'value'\n    temp.reset_index(inplace=True)\n    plt.bar(temp.index, temp['count'], alpha=0.5, label='target=1')\n    \n    # Set x-axis label\n    plt.xlabel(f)\n    \n    # Add legend to the plot\n    plt.legend()\n    \n    # Set x-axis ticks to match the values\n    plt.xticks(temp.index, temp.value)\n\n# Add overall title to the plot\nplt.suptitle('Categorical Features', fontsize=20, y=0.93)\n\n# Show the plot\nplt.show()\n\n# Delete the temporary DataFrame 'temp'\ndel temp\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:40.474915Z","iopub.execute_input":"2024-04-15T14:16:40.475462Z","iopub.status.idle":"2024-04-15T14:16:44.549318Z","shell.execute_reply.started":"2024-04-15T14:16:40.475399Z","shell.execute_reply":"2024-04-15T14:16:44.547780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n## Observations:\nEvery categorical attribute consists of a maximum of eight distinct categories, enabling the possibility of employing One-hot encoding. I intend to examine both one-hot encoding and ordinal encoding methodologies in my modeling notebook.\n\nThe disparities in distributions between target=0 and target=1 suggest that categorical attributes offer predictive insights into the target variable. Hence, it is advisable to explore the modeling of these categorical variables.\n\nThe attributes D_114, D_116, D_120, and D_66 are binary in nature, with values restricted to 0, 1, or left as missing.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract columns starting with 'D_' representing delinquency features from the train_data\nD_cols = train_data.columns[pd.Series(train_data.columns).str.startswith('D_')]\n# Extract columns starting with 'B_' representing balance features from the train_data\nB_cols = train_data.columns[pd.Series(train_data.columns).str.startswith('B_')]\n# Extract columns starting with 'S_' representing spending features from the train_data\nS_cols = train_data.columns[pd.Series(train_data.columns).str.startswith('S_')]\n# Extract columns starting with 'R_' representing risk features from the train_data\nR_cols = train_data.columns[pd.Series(train_data.columns).str.startswith('R_')]\n# Extract columns starting with 'P_' representing payment features from the train_data\nP_cols = train_data.columns[pd.Series(train_data.columns).str.startswith('P_')]\n\n# Create a dictionary to map feature category names to their counts\nDict = {\n    'Delinquency': len(D_cols), \n    'Spend': len(S_cols), \n    'Payment': len(P_cols), \n    'Balance': len(B_cols), \n    'Risk': len(R_cols),\n}\n\n# Set up the figure size for the bar plot\nplt.figure(figsize=(10, 5))\n# Create a bar plot to visualize the number of columns in each category\nsns.barplot(x=list(Dict.keys()), y=list(Dict.values()))\n# Add a legend to the plot\nplt.legend();\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:44.551589Z","iopub.execute_input":"2024-04-15T14:16:44.552367Z","iopub.status.idle":"2024-04-15T14:16:44.844303Z","shell.execute_reply.started":"2024-04-15T14:16:44.552300Z","shell.execute_reply":"2024-04-15T14:16:44.842993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualization of Numerical Features\n\n\n\n #### Delinquency Variables","metadata":{}},{"cell_type":"code","source":"# Calculate and print the number of delinquency features in the dataset\nprint(f\"Number of delinquency features: {len([f for f in train_data.columns if (f not in  ['customer_ID', 'target', 'S_2']) and (f.startswith(('D')))])}\")\n\n# Extract and sort the numerical delinquency features from the dataset\ndelin_features = sorted([f for f in train_data.columns if (f not in cat_features + ['customer_ID', 'target', 'S_2']) and (f.startswith(('D')))])\n\n# Calculate and print the number of numerical delinquency features\nprint(f\"Number of numerical delinquency features: {len(delin_features)}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:44.846009Z","iopub.execute_input":"2024-04-15T14:16:44.846507Z","iopub.status.idle":"2024-04-15T14:16:44.853914Z","shell.execute_reply.started":"2024-04-15T14:16:44.846464Z","shell.execute_reply":"2024-04-15T14:16:44.853090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist(features, title):\n    # Create a figure and an array of subplots based on the number of features\n    fig, axs = plt.subplots(int(len(features)/4)+(1 if len(features)%4!=0 else 0),4, figsize=(15,3*(int(len(features)/4)+(1 if len(features)%4!=0 else 0))))\n\n    # Set the main title for the entire plot\n    plt.suptitle(f'\\n{title}', fontsize=20, y=1)\n\n    # Iterate through each feature and plot its histogram\n    for i, f in enumerate(features):\n        # Determine the current subplot for the feature\n        ax = axs[int(i/4), i%4] if ((int(len(features)/4)+(1 if len(features)%4!=0 else 0))>1) else axs[i%4]\n        \n        # Plot the histogram for the current feature\n        ax.hist(train_data[f], bins=200)\n        \n        # Set the title of the subplot to the feature name\n        ax.set_title(f)\n\n        # Create a smaller inset subplot within the larger subplot\n        inset_ax = ax.inset_axes([0.6, 0.6, 0.35, 0.35])  # [left, bottom, width, height]\n        inset_ax.hist(train_data[f], bins=200)\n        inset_ax.set_ylim([0,0.001])\n        inset_ax.set_yticks([])\n        \n        # Hide the y-axis labels for the main subplot\n        ax.set_yticklabels([])\n\n    # Adjust the layout of subplots to prevent overlap\n    plt.tight_layout()\n    \n    # Display the plot\n    plt.show()\n\n# Call the function to plot histograms for delinquency features\nplot_hist(delin_features, 'Delinquency Features')\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:16:44.854954Z","iopub.execute_input":"2024-04-15T14:16:44.855289Z","iopub.status.idle":"2024-04-15T14:19:08.731080Z","shell.execute_reply.started":"2024-04-15T14:16:44.855263Z","shell.execute_reply":"2024-04-15T14:19:08.729791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Drawing from the histogram plots, certain attributes like D_103 initially exhibit a discrete pattern. However, upon closer inspection of the plots, it becomes apparent that there is additional noise overlaying the values. For instance, D_103 spans the range [0, 0.01] and [1, 1.01]. This phenomenon might be attributed to the incorporation of noise into the data, potentially as a means of anonymizing customer information by AMEX.\n\n\n\nSome of the plots have white space at the left or right end. This may show that there aer some outliers or rare events in the data.\n\nIt is possible to remove this added noise. This will likely result in higher model accuracy. We can use a transformatiom such as: train_data['D_103'] = train_data['D_103'].apply(lambda t: np.floor(t))\n\n\n\nIf the two distributions (default vs. paid) are very separate for a feature, it shows that the feature could have useful information about the target and should be used in predictive modelling.\n\nFor some of the features such as D_134, D_121, etc., the distribution for bad customers is similar in shape to the distribution of good customers, but the distribution is shifted up or down in the y axis. This may show similar overall behavioral pattern for good and bad customers. The shift in the scale of the distribution could show that the scale of the behavior (such as amout spent) is different for the two groups.","metadata":{}},{"cell_type":"markdown","source":"##  Relationship Among Delinquency Features","metadata":{}},{"cell_type":"code","source":"def plot_corr_heatmap(features, title):\n    # Calculate the correlation matrix for the provided features\n    corr = last_statement_df[features].iloc[:,:-1].corr()\n\n    # Create a mask to hide the upper triangle of the correlation matrix\n    mask = np.triu(np.ones_like(corr, dtype=bool))[1:,:-1]\n    \n    # Extract the lower triangle of the correlation matrix\n    corr = corr.iloc[1:,:-1].copy()\n\n    # Create a figure and axis for the heatmap\n    fig, ax = plt.subplots(figsize=(48,48))\n    \n    # Plot the heatmap with correlation values and annotations\n    sns.heatmap(corr, mask=mask, vmin=-1, vmax=1, center=0, annot=True, fmt='.2f', cmap='viridis', annot_kws={'fontsize':10,'fontweight':'bold'}, cbar=False)\n\n    # Remove ticks from both x and y axes\n    ax.tick_params(left=False, bottom=False)\n    \n    # Rotate and align x-axis tick labels for better readability\n    ax.set_xticklabels(ax.get_xticklabels(), rotation=45, horizontalalignment='right', fontsize=12)\n    \n    # Set font size for y-axis tick labels\n    ax.set_yticklabels(ax.get_yticklabels(), fontsize=12)\n\n    # Set the title for the heatmap\n    plt.title(f'{title}\\n', fontsize=26)\n    \n    # Display the plot\n    fig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:08.732931Z","iopub.execute_input":"2024-04-15T14:19:08.733501Z","iopub.status.idle":"2024-04-15T14:19:08.744331Z","shell.execute_reply.started":"2024-04-15T14:19:08.733459Z","shell.execute_reply":"2024-04-15T14:19:08.743363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_statement_df = train_data.groupby('customer_ID').tail(1).set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:08.745713Z","iopub.execute_input":"2024-04-15T14:19:08.746471Z","iopub.status.idle":"2024-04-15T14:19:11.942946Z","shell.execute_reply.started":"2024-04-15T14:19:08.746439Z","shell.execute_reply":"2024-04-15T14:19:11.941713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols=[col for col in last_statement_df.columns if (col.startswith(('D','t'))) & (col not in cat_features)]","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:11.944543Z","iopub.execute_input":"2024-04-15T14:19:11.944912Z","iopub.status.idle":"2024-04-15T14:19:11.951178Z","shell.execute_reply.started":"2024-04-15T14:19:11.944884Z","shell.execute_reply":"2024-04-15T14:19:11.949756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_corr_heatmap(cols, 'Correlation Between Delinquency Variables')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:11.953436Z","iopub.execute_input":"2024-04-15T14:19:11.954606Z","iopub.status.idle":"2024-04-15T14:19:35.617287Z","shell.execute_reply.started":"2024-04-15T14:19:11.954557Z","shell.execute_reply":"2024-04-15T14:19:35.616201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nSeveral of the delinquency variables exhibit strong correlations.","metadata":{}},{"cell_type":"code","source":"# Select columns that start with 'D' or 't' and are not in the list of categorical features\ncols = [col for col in last_statement_df.columns if (col.startswith(('D', 't'))) and (col not in cat_features)]\n\n# Calculate the correlation matrix for the selected columns\ncorr = last_statement_df[cols].iloc[:, :-1].corr()\n\n# Fill the diagonal of the correlation matrix with 0.0\nnp.fill_diagonal(corr.values, 0.0)\n\n# Create a mask to hide the upper triangle of the correlation matrix\nmask = np.triu(np.ones_like(corr, dtype=bool))\n\n# Mask the upper triangle of the correlation matrix\ncorr = corr.mask(mask)\n\n# Find pairs of highly correlated features (correlation coefficient > 0.9)\nhighly_correlated_features = corr[np.abs(corr) > 0.9].stack().reset_index()\n\n# Sort the pairs by the absolute correlation coefficient in descending order\nhighly_correlated_features = highly_correlated_features.iloc[abs(highly_correlated_features[0]).argsort()[::-1]].reset_index()\n\n# Display the highly correlated features\nhighly_correlated_features\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:35.618873Z","iopub.execute_input":"2024-04-15T14:19:35.619689Z","iopub.status.idle":"2024-04-15T14:19:44.378806Z","shell.execute_reply.started":"2024-04-15T14:19:35.619647Z","shell.execute_reply":"2024-04-15T14:19:44.377653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_highly_correlated_features(features, title, highly_correlated_features):\n    # Create a figure and an array of subplots based on the number of highly correlated features\n    fig, axs = plt.subplots(int(len(highly_correlated_features)/4)+(1 if len(highly_correlated_features)%4!=0 else 0), 4, figsize=(16,5*(int(len(highly_correlated_features)/4)+(1 if len(highly_correlated_features)%4!=0 else 0))))\n    \n    # Initialize index variable\n    i = 0\n    \n    # Iterate through pairs of highly correlated features\n    for f, g in zip(highly_correlated_features.level_0, highly_correlated_features.level_1):\n        # Determine the current subplot for the feature pair\n        ax = axs[int(i/4), i%4] if ((int(len(highly_correlated_features)/4)+(1 if len(highly_correlated_features)%4!=0 else 0))>1) else axs[i%4]\n        \n        # Plot scatter plot for the feature pair if it's not the 8th pair\n        if(i != 7):\n            ax.hexbin(x=f, y=g, data=last_statement_df[features], bins='log', gridsize=40, cmap='inferno')\n            ax.set(xlabel=f, ylabel=g)\n            ax.text(0.05, 0.9, 'Correlation: {:.4f}'.format(highly_correlated_features[0][i]), transform=ax.transAxes, bbox=dict(boxstyle=\"round,pad=0.3\", fc=\"white\"))\n            ax.tick_params(left=False, bottom=False)\n        else:\n            # Plot scatter plot for the 8th pair (which is not hexbin)\n            ax.plot(last_statement_df[features][f], last_statement_df[features][g], '.')\n            ax.set(xlabel=f, ylabel=g)\n            ax.text(0.05, 0.9, 'Correlation: {:.4f}'.format(highly_correlated_features[0][i]), transform=ax.transAxes, bbox=dict(boxstyle=\"round,pad=0.3\", fc=\"white\"))\n            ax.tick_params(left=False, bottom=False)\n\n        # Increment index\n        i += 1\n\n    # Remove spines from the plot\n    sns.despine()\n    \n    # Set the main title for the entire plot\n    fig.suptitle(f'\\n{title})', fontsize=14)\n    \n    # Show the plot\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:44.380882Z","iopub.execute_input":"2024-04-15T14:19:44.381811Z","iopub.status.idle":"2024-04-15T14:19:44.397041Z","shell.execute_reply.started":"2024-04-15T14:19:44.381763Z","shell.execute_reply":"2024-04-15T14:19:44.395877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_highly_correlated_featuers(features, title, highly_correlated_features):\n    # Calculate the number of subplots needed based on the count of highly correlated features. Arrange subplots in a grid that has 4 columns.\n    # The grid size adjusts based on the number of pairs to ensure all plots fit.\n    fig, axs = plt.subplots(int(len(highly_correlated_features)/4)+(1 if len(highly_correlated_features)%4!=0 else 0),4, figsize=(16,5*(int(len(highly_correlated_features)/4)+(1 if len(highly_correlated_features)%4!=0 else 0))))\n    i=0\n    for f,g in zip(highly_correlated_features.level_0, highly_correlated_features.level_1):\n\n        # Determine the correct subplot position for the current feature pair.\n        # Adjusts for cases with a single row or multiple rows of subplots.\n        ax = axs[int(i/4), i%4] if ((int(len(highly_correlated_features)/4)+(1 if len(highly_correlated_features)%4!=0 else 0))>1) else axs[i%4]\n         \n        # Special case for a specific subplot (e.g., the eighth one, index 7) to use a dot plot instead of a hexbin plot.\n        if(i==7):\n            ax.plot(last_statement_df[features][f], last_statement_df[features][g], '.')\n            ax.set(xlabel=f,ylabel=g)\n            ax.text(0.05, 0.9, 'Correlation: {:.4f}'.format(highly_correlated_features[0][i]), transform=ax.transAxes, bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\n            ax.tick_params(left=False,bottom=False)\n\n            i = i + 1\n            continue\n        \n\n     # For other plots, use a hexbin plot to show the density of points with logarithmic binning.\n        ax.hexbin(x=f, y=g, data=last_statement_df[features], bins='log', gridsize=40, cmap='inferno')\n        ax.set(xlabel=f,ylabel=g)\n\n        ax.text(0.05, 0.9, 'Correlation: {:.4f}'.format(highly_correlated_features[0][i]), transform=ax.transAxes, bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\n        # Remove ticks for a cleaner look.\n        ax.tick_params(left=False,bottom=False)\n        i = i + 1\n\n    sns.despine()  # Remove the top and right spines from all plots for a neater appearance.\n    fig.suptitle(f'\\n{title})',fontsize=14)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:44.398796Z","iopub.execute_input":"2024-04-15T14:19:44.399135Z","iopub.status.idle":"2024-04-15T14:19:44.415387Z","shell.execute_reply.started":"2024-04-15T14:19:44.399109Z","shell.execute_reply":"2024-04-15T14:19:44.414150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_highly_correlated_featuers(cols, 'Most Highly-Correlated Delinquency Variables (Log Transformed Relationship)', highly_correlated_features)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:44.420513Z","iopub.execute_input":"2024-04-15T14:19:44.421981Z","iopub.status.idle":"2024-04-15T14:19:48.941900Z","shell.execute_reply.started":"2024-04-15T14:19:44.421925Z","shell.execute_reply":"2024-04-15T14:19:48.940552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Several Delinquency variables display strong correlations, with the highest correlation coefficient reaching 0.999824, observed between D_77 and D_62.\n\nThere are some missing correlations evident in the correlation heatmap, which can be attributed to null values present in the dataset.","metadata":{}},{"cell_type":"markdown","source":"#### Spend Variables","metadata":{}},{"cell_type":"code","source":"spend_features = sorted([f for f in train_data.columns if (f not in ['customer_ID', 'target', 'S_2']) and (f.startswith(('S')))])\nprint(f\"Number of spend features: {len(spend_features)}\")\nprint(f\"Number of numerical spend features: {len([f for f in spend_features if f not in cat_features])}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:48.943379Z","iopub.execute_input":"2024-04-15T14:19:48.943741Z","iopub.status.idle":"2024-04-15T14:19:48.952501Z","shell.execute_reply.started":"2024-04-15T14:19:48.943713Z","shell.execute_reply":"2024-04-15T14:19:48.951048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist(spend_features, 'Spend Features')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:19:48.954320Z","iopub.execute_input":"2024-04-15T14:19:48.954736Z","iopub.status.idle":"2024-04-15T14:20:27.543940Z","shell.execute_reply.started":"2024-04-15T14:19:48.954706Z","shell.execute_reply":"2024-04-15T14:20:27.542510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_corr_heatmap(cols, 'Correlations Between Spend Variables')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:20:27.545762Z","iopub.execute_input":"2024-04-15T14:20:27.546190Z","iopub.status.idle":"2024-04-15T14:20:49.666483Z","shell.execute_reply.started":"2024-04-15T14:20:27.546152Z","shell.execute_reply":"2024-04-15T14:20:49.664456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix for the selected columns\ncorr = last_statement_df[cols].iloc[:,:-1].corr()\n\n# Fill the diagonal of the correlation matrix with 0.0 to avoid self-correlation\nnp.fill_diagonal(corr.values, 0.0)\n\n# Create a mask to hide the upper triangle of the correlation matrix\nmask = np.triu(np.ones_like(corr, dtype=bool))\n\n# Mask the upper triangle of the correlation matrix\ncorr = corr.mask(mask)\n\n# Find pairs of highly correlated features (correlation coefficient > 0.9)\nhighly_correlated_features = corr[np.abs(corr) > 0.9].stack().reset_index()\n\n# Sort the pairs by the absolute correlation coefficient in descending order\nhighly_correlated_features = highly_correlated_features.iloc[abs(highly_correlated_features[0]).argsort()[::-1]].reset_index()\n\n# Display the highly correlated features\nhighly_correlated_features\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:20:49.674140Z","iopub.execute_input":"2024-04-15T14:20:49.675383Z","iopub.status.idle":"2024-04-15T14:20:58.284200Z","shell.execute_reply.started":"2024-04-15T14:20:49.675282Z","shell.execute_reply":"2024-04-15T14:20:58.282955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_highly_correlated_featuers(cols, 'Most Highly-Correlated Spend Variables (Log Transformed Relationship)', highly_correlated_features)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:20:58.285865Z","iopub.execute_input":"2024-04-15T14:20:58.286976Z","iopub.status.idle":"2024-04-15T14:21:02.776957Z","shell.execute_reply.started":"2024-04-15T14:20:58.286930Z","shell.execute_reply":"2024-04-15T14:21:02.775684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The hist plots illustrate noticeable disparities in distributions between target values of 0 and 1. It appears that the spend variables carry significant information regarding the target and warrant inclusion in modeling efforts.\n\nSeveral spend features exhibit strong correlations, notably S_22 and S_24 with a Pearson correlation coefficient of 0.965. It's important to note that Pearson's correlation coefficient solely evaluates linear relationships, potentially overlooking nonlinear associations.","metadata":{}},{"cell_type":"markdown","source":"#### Payment Variables","metadata":{}},{"cell_type":"code","source":"payment_features = sorted([f for f in train_data.columns if (f not in ['customer_ID', 'target', 'S_2']) and (f.startswith(('P')))])\nprint(f\"Number of payment features: {len(payment_features)}\")\nprint(f\"Number of numerical payment features: {len([f for f in payment_features if f not in cat_features])}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:21:02.778525Z","iopub.execute_input":"2024-04-15T14:21:02.778916Z","iopub.status.idle":"2024-04-15T14:21:02.786115Z","shell.execute_reply.started":"2024-04-15T14:21:02.778882Z","shell.execute_reply":"2024-04-15T14:21:02.784734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist(payment_features, 'Payment Features')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:21:02.787707Z","iopub.execute_input":"2024-04-15T14:21:02.788150Z","iopub.status.idle":"2024-04-15T14:21:08.485884Z","shell.execute_reply.started":"2024-04-15T14:21:02.788113Z","shell.execute_reply":"2024-04-15T14:21:08.484356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_corr_heatmap(cols, 'Correlations Between Payment Variables')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:21:08.487900Z","iopub.execute_input":"2024-04-15T14:21:08.488441Z","iopub.status.idle":"2024-04-15T14:21:29.672324Z","shell.execute_reply.started":"2024-04-15T14:21:08.488388Z","shell.execute_reply":"2024-04-15T14:21:29.670686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nThere are just three payment factors, and they don't show strong correlation with each other.\n\nAccording to the KDE plot, P_2 appears to provide the most relevant information about the target.","metadata":{}},{"cell_type":"markdown","source":"#### Balance Variables","metadata":{}},{"cell_type":"code","source":"balance_features = sorted([f for f in train_data.columns if (f not in  ['customer_ID', 'target', 'S_2']) and (f.startswith(('B')))])\nprint(f\"Number of balance features: {len(balance_features)}\")\nprint(f\"Number of numerical balance features: {len([f for f in balance_features if f not in cat_features])}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:21:29.674136Z","iopub.execute_input":"2024-04-15T14:21:29.674607Z","iopub.status.idle":"2024-04-15T14:21:29.683175Z","shell.execute_reply.started":"2024-04-15T14:21:29.674568Z","shell.execute_reply":"2024-04-15T14:21:29.682027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist(balance_features, 'Balance Features')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:21:29.684890Z","iopub.execute_input":"2024-04-15T14:21:29.685209Z","iopub.status.idle":"2024-04-15T14:22:41.539626Z","shell.execute_reply.started":"2024-04-15T14:21:29.685184Z","shell.execute_reply":"2024-04-15T14:22:41.538379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_corr_heatmap(cols, 'Correlations Between Balance Variables')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:22:41.541283Z","iopub.execute_input":"2024-04-15T14:22:41.542370Z","iopub.status.idle":"2024-04-15T14:23:03.596949Z","shell.execute_reply.started":"2024-04-15T14:22:41.542326Z","shell.execute_reply":"2024-04-15T14:23:03.595364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nThere are 38 numerical balance features and 2 categorical balance features, with some showing significant correlations. These features contain valuable information related to the target variable and will be incorporated into the modeling process. Upon examination of the hist plots, it appears that certain variables exhibit outlier values.","metadata":{}},{"cell_type":"markdown","source":"#### Risk Variables","metadata":{}},{"cell_type":"code","source":"risk_features = sorted([f for f in train_data.columns if (f not in ['customer_ID', 'target', 'S_2']) and (f.startswith(('R')))])\nprint(f\"Number of risk features: {len(risk_features)}\")\nprint(f\"Number of numerical risk features: {len([f for f in risk_features if f not in cat_features])}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:23:03.598571Z","iopub.execute_input":"2024-04-15T14:23:03.599157Z","iopub.status.idle":"2024-04-15T14:23:03.604853Z","shell.execute_reply.started":"2024-04-15T14:23:03.599119Z","shell.execute_reply":"2024-04-15T14:23:03.603983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist(risk_features, 'Risk Features')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:23:03.606301Z","iopub.execute_input":"2024-04-15T14:23:03.606903Z","iopub.status.idle":"2024-04-15T14:23:54.346342Z","shell.execute_reply.started":"2024-04-15T14:23:03.606859Z","shell.execute_reply":"2024-04-15T14:23:54.345155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_corr_heatmap(cols, 'Correlations Between Risk Variables')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:23:54.348072Z","iopub.execute_input":"2024-04-15T14:23:54.348463Z","iopub.status.idle":"2024-04-15T14:24:16.671680Z","shell.execute_reply.started":"2024-04-15T14:23:54.348430Z","shell.execute_reply":"2024-04-15T14:24:16.669631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 28 numerical risk attributes and zero categorical ones in the dataset.\n\nMost risk factors exhibit low levels of correlation with each other, with the exception of R_5 and R_8, which show the highest correlation.\n\nThese variables hold valuable information about the target and will be integrated into the modeling procedure with care.","metadata":{}},{"cell_type":"markdown","source":"## Relationship between Features and the Target","metadata":{}},{"cell_type":"code","source":"# Selecting numerical columns from the training data, excluding the 'target' column,\n# and computing correlation with the 'target' column.\ntarg = train_data.select_dtypes(include='number').drop([\"target\"], axis=1).corrwith(train_data['target'], axis=0)\nval = [str(round(v ,1) *100) + '%' for v in targ.values]\n\n# Formatting the correlation values to percentage strings rounded to 1 decimal place.\nfig, ax = plt.subplots(figsize=(14, 25))\nbars = ax.barh(targ.index, targ.values, color=\"lightblue\")\n\nfor bar, label in zip(bars, val):\n    ax.text(bar.get_width(), bar.get_y() + bar.get_height() / 2, label,ha='left', va='center', fontsize=12)\n\n\n# Annotating each bar with its corresponding percentage value.\n# The loop iterates over the bars and their labels, positioning the label\n# according to the bar's width and height for proper alignment.\nax.set_title(\"Correlation of Numerical Features with the Target\", fontsize=16)\nax.set_facecolor('none')\nax.tick_params(axis='y', which='both', left=False)\n\nax.set_xticks([])\nax.set_xticklabels([])\n\nax.spines[['top', 'bottom', 'left', 'right']].set_visible(False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:24:16.673755Z","iopub.execute_input":"2024-04-15T14:24:16.674258Z","iopub.status.idle":"2024-04-15T14:25:05.765838Z","shell.execute_reply.started":"2024-04-15T14:24:16.674214Z","shell.execute_reply":"2024-04-15T14:25:05.764396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nExamining the relationship between features and the target variable reveals correlations spanning from -61% to 50%.\n\nP_2 exhibits the strongest negative correlation with the target at -61%.\n\nExploring the utilization of the most highly correlated features in modeling could yield promising results. .","metadata":{}},{"cell_type":"markdown","source":"## Model Development\n","metadata":{}},{"cell_type":"markdown","source":"\nDue to the dataset's substantial size, I'm constrained by memory limitations and can only work with a subset of 10,000 datasets. \n\nContaining 5.5 million rows, the dataset exceeds my available memory capacity during model training. Consequently, I'm restricting my analysis to the first 10,000 rows of data.\n\nFor my analysis, I'll be focusing solely on the numerical features, opting to drop the 11 categorical features.","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_feather(\"../input/amexfeather/train_data.ftr\").iloc[:10000,:]\ncategorical_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\ntrain_data = train_data.drop(columns = categorical_features)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:05.767437Z","iopub.execute_input":"2024-04-15T14:25:05.767788Z","iopub.status.idle":"2024-04-15T14:25:15.226267Z","shell.execute_reply.started":"2024-04-15T14:25:05.767757Z","shell.execute_reply":"2024-04-15T14:25:15.225119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:15.227977Z","iopub.execute_input":"2024-04-15T14:25:15.228818Z","iopub.status.idle":"2024-04-15T14:25:15.256379Z","shell.execute_reply.started":"2024-04-15T14:25:15.228782Z","shell.execute_reply":"2024-04-15T14:25:15.255089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.drop([\"customer_ID\",\"S_2\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:15.258508Z","iopub.execute_input":"2024-04-15T14:25:15.259001Z","iopub.status.idle":"2024-04-15T14:25:15.268672Z","shell.execute_reply.started":"2024-04-15T14:25:15.258957Z","shell.execute_reply":"2024-04-15T14:25:15.267273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:15.270421Z","iopub.execute_input":"2024-04-15T14:25:15.270888Z","iopub.status.idle":"2024-04-15T14:25:15.292482Z","shell.execute_reply.started":"2024-04-15T14:25:15.270851Z","shell.execute_reply":"2024-04-15T14:25:15.291333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is a significant number of missing data. It is not reasonable to drop all columns or rows that have a missing value.\n\nFrom my earlier histplots, we can infer that most features are not normaly distributed. It is either negatively skewed, positively skewed or no skew at all. With that said,it is difficult to identify the points that the features are centered, and to deal with the null values it makes sense to fill na with 0","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:15.294133Z","iopub.execute_input":"2024-04-15T14:25:15.295030Z","iopub.status.idle":"2024-04-15T14:25:15.319490Z","shell.execute_reply.started":"2024-04-15T14:25:15.294987Z","shell.execute_reply":"2024-04-15T14:25:15.318262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe().T","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:15.321348Z","iopub.execute_input":"2024-04-15T14:25:15.321824Z","iopub.status.idle":"2024-04-15T14:25:15.908840Z","shell.execute_reply.started":"2024-04-15T14:25:15.321782Z","shell.execute_reply":"2024-04-15T14:25:15.907421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The variation in ranges among the features will pose challenges for some of my ML models. Typically, it's advantageous to normalize input values to mitigate this issue. Data normalization is a vital preprocessing step in model development, promoting stable and efficient training, accelerated convergence, and enhanced generalization.\n\n","metadata":{}},{"cell_type":"code","source":"train_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:15.910782Z","iopub.execute_input":"2024-04-15T14:25:15.911558Z","iopub.status.idle":"2024-04-15T14:25:15.949518Z","shell.execute_reply.started":"2024-04-15T14:25:15.911506Z","shell.execute_reply":"2024-04-15T14:25:15.946001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[\"target\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:15.951403Z","iopub.execute_input":"2024-04-15T14:25:15.951813Z","iopub.status.idle":"2024-04-15T14:25:15.963706Z","shell.execute_reply.started":"2024-04-15T14:25:15.951777Z","shell.execute_reply":"2024-04-15T14:25:15.962407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X= train_data.drop(\"target\", axis=1)\ny= train_data[\"target\"]\n\n\nfrom sklearn.preprocessing import StandardScaler\n\n# Initialize the scaler\nscaler = StandardScaler()\n\n# Fit and transform the data\nX_s = scaler.fit_transform(X)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:15.966141Z","iopub.execute_input":"2024-04-15T14:25:15.966630Z","iopub.status.idle":"2024-04-15T14:25:16.393991Z","shell.execute_reply.started":"2024-04-15T14:25:15.966587Z","shell.execute_reply":"2024-04-15T14:25:16.392713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"75% of the data will be used to train the model.\n\n25% of the data will be used for validation.","metadata":{}},{"cell_type":"code","source":"x_train,x_test,y_train,y_test= train_test_split(X_s,y, train_size=0.75, random_state=20)","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:16.395823Z","iopub.execute_input":"2024-04-15T14:25:16.396342Z","iopub.status.idle":"2024-04-15T14:25:16.530849Z","shell.execute_reply.started":"2024-04-15T14:25:16.396274Z","shell.execute_reply":"2024-04-15T14:25:16.529751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score, StratifiedKFold\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom lightgbm import LGBMClassifier","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:16.532737Z","iopub.execute_input":"2024-04-15T14:25:16.533286Z","iopub.status.idle":"2024-04-15T14:25:17.559246Z","shell.execute_reply.started":"2024-04-15T14:25:16.533236Z","shell.execute_reply":"2024-04-15T14:25:17.558096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {\n    'Logistic Regression': LogisticRegression(max_iter=1000)\n,\n    'Support Vector Classifier': SVC(),\n    'Decision Tree Classifier': DecisionTreeClassifier(),\n    'Light GBM Classifier': LGBMClassifier()\n}\n\nresults = []\n\nfor model_name, model in models.items():\n    kfold = StratifiedKFold(n_splits=10, shuffle=True, random_state=42)\n    cv_scores = cross_val_score(model, x_train, y_train, cv=kfold, scoring='accuracy')\n    results.append([model_name, cv_scores.mean()])\n\nresults_df = pd.DataFrame(results, columns=['Model', 'Accuracy'])\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:25:17.560875Z","iopub.execute_input":"2024-04-15T14:25:17.561755Z","iopub.status.idle":"2024-04-15T14:26:32.095547Z","shell.execute_reply.started":"2024-04-15T14:25:17.561707Z","shell.execute_reply":"2024-04-15T14:26:32.094479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_df.index = range(1, 5)\n\nresults_df","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:26:32.097292Z","iopub.execute_input":"2024-04-15T14:26:32.097781Z","iopub.status.idle":"2024-04-15T14:26:32.110870Z","shell.execute_reply.started":"2024-04-15T14:26:32.097747Z","shell.execute_reply":"2024-04-15T14:26:32.109632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\n\n# Training each model on the full training dataset and evaluating on the test dataset\ntest_results = []\n\nfor model_name, model in models.items():\n    # Fit model\n    model.fit(x_train, y_train)\n    # Predict on test data\n    y_pred = model.predict(x_test)\n    # Calculate accuracy\n    accuracy = accuracy_score(y_test, y_pred)\n    # Append results\n    test_results.append([model_name, accuracy])\n\n# Creating a DataFrame to display the results\ntest_results_df = pd.DataFrame(test_results, columns=['Model', 'Test Accuracy'])\n\n# Print or return the DataFrame\nprint(test_results_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:36:50.615989Z","iopub.execute_input":"2024-04-15T14:36:50.616791Z","iopub.status.idle":"2024-04-15T14:37:15.304976Z","shell.execute_reply.started":"2024-04-15T14:36:50.616746Z","shell.execute_reply":"2024-04-15T14:37:15.303856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.bar(results_df['Model'], results_df['Accuracy'], color='skyblue')\nplt.xlabel('Model')\nplt.ylabel('Accuracy')\nplt.title('Accuracy of Models')\nplt.xticks(rotation=45, ha='right')  # Rotate x-axis labels for better readability\nplt.tight_layout()  # Adjust layout to prevent clipping of labels\nplt.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:26:32.112591Z","iopub.execute_input":"2024-04-15T14:26:32.112951Z","iopub.status.idle":"2024-04-15T14:26:32.443851Z","shell.execute_reply.started":"2024-04-15T14:26:32.112923Z","shell.execute_reply":"2024-04-15T14:26:32.442585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Placeholder for feature importances\nfeature_importances = {}\n\n# Extracting feature importances\nfor model_name, model in models.items():\n    # Checking if model has feature_importances_ attribute (Decision Tree, LightGBM)\n    if hasattr(model, 'feature_importances_'):\n        feature_importances[model_name] = model.feature_importances_\n    # For Logistic Regression, using the coefficients\n    elif model_name == 'Logistic Regression' and hasattr(model, 'coef_'):\n        feature_importances[model_name] = model.coef_[0]\n    # Note: SVC does not provide a direct way for feature importance for non-linear kernels\n\n# Displaying feature importances\nfor model_name, importances in feature_importances.items():\n    print(f\"\\nFeature Importances for {model_name}:\")\n    for i, v in enumerate(importances):\n        print(f'Feature: {i}, Score: {v:.5f}')","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:34:55.400215Z","iopub.execute_input":"2024-04-15T14:34:55.400686Z","iopub.status.idle":"2024-04-15T14:34:55.413341Z","shell.execute_reply.started":"2024-04-15T14:34:55.400651Z","shell.execute_reply":"2024-04-15T14:34:55.412147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfeature_names = list(X.columns) \n\n# Placeholder for top 5 feature importances for each model\ntop_features = {}\n\n# Extract and sort top 5 feature importances\nfor model_name, model in models.items():\n    if hasattr(model, 'feature_importances_'):\n        importances = pd.Series(model.feature_importances_, index=feature_names).sort_values(ascending=False)\n        top_features[model_name] = importances.head(5)\n    elif model_name == 'Logistic Regression' and hasattr(model, 'coef_'):\n        importances = pd.Series(model.coef_[0], index=feature_names).abs().sort_values(ascending=False)\n        top_features[model_name] = importances.head(5)\n\n# Convert top features to DataFrame for table format\ntop_features_df_list = []\n\nfor model_name, features in top_features.items():\n    df = features.reset_index()\n    df.columns = ['Feature', 'Importance']\n    df['Model'] = model_name\n    top_features_df_list.append(df)\n\n# Concatenate all model's top features into a single DataFrame\nfinal_df = pd.concat(top_features_df_list, ignore_index=True)\n\n# Pivot the DataFrame for better readability\npivot_df = final_df.pivot(index='Feature', columns='Model', values='Importance').fillna(0)\n\npivot_df\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:37:15.307223Z","iopub.execute_input":"2024-04-15T14:37:15.307925Z","iopub.status.idle":"2024-04-15T14:37:15.364682Z","shell.execute_reply.started":"2024-04-15T14:37:15.307884Z","shell.execute_reply":"2024-04-15T14:37:15.362970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, precision_recall_curve, auc, accuracy_score, precision_score, recall_score, f1_score\nimport numpy as np\n\n# Define different train-test splits\ntrain_test_splits = [(0.6, 0.4), (0.7, 0.3), (0.8, 0.2)]\n\n# Define models\nmodels = {\n    'Logistic Regression': LogisticRegression(solver='liblinear',max_iter=1000),\n    'Support Vector Classifier': SVC(probability=True),\n    'Decision Tree Classifier': DecisionTreeClassifier(),\n    'Light GBM Classifier': LGBMClassifier()\n}\n\n# Initialize lists to store results\nroc_auc_results = {}\nprecision_recall_results = {}\nmetrics_results = []\n\n# Iterate over train-test splits\nfor train_size, test_size in train_test_splits:\n    x_train, x_test, y_train, y_test = train_test_split(X_s, y, train_size=train_size, test_size=test_size, random_state=20)\n    \n    # Iterate over models\n    for model_name, model in models.items():\n        # Train model\n        model.fit(x_train, y_train)\n        \n        # Predict probabilities\n        y_pred_proba = model.predict_proba(x_test)[:, 1]\n        \n        # ROC curve\n        fpr, tpr, _ = roc_curve(y_test, y_pred_proba)\n        roc_auc = auc(fpr, tpr)\n        roc_auc_results[(train_size, test_size, model_name)] = roc_auc\n        \n        # Precision-recall curve\n        precision, recall, _ = precision_recall_curve(y_test, y_pred_proba)\n        precision_recall_results[(train_size, test_size, model_name)] = (precision, recall)\n        \n        # Metrics\n        y_pred = model.predict(x_test)\n        accuracy = accuracy_score(y_test, y_pred)\n        precision = precision_score(y_test, y_pred)\n        recall = recall_score(y_test, y_pred)\n        f1 = f1_score(y_test, y_pred)\n        metrics_results.append((train_size, test_size, model_name, accuracy, precision, recall, f1))\n\n# Plot ROC curve and precision-recall curve for each model\n# Plot ROC curve and precision-recall curve for each model\nfor model_name in models.keys():\n    plt.figure(figsize=(10, 4))\n    plt.subplot(1, 2, 1)\n    plt.title(f'ROC Curve - {model_name}')\n    for (train_size, test_size, model_name_key), roc_auc in roc_auc_results.items():\n        if model_name == model_name_key:\n            plt.plot(fpr, tpr, label=f'Train: {train_size}, Test: {test_size} (AUC = {roc_auc:.2f})')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.legend()\n    \n    plt.subplot(1, 2, 2)\n    plt.title(f'Precision-Recall Curve - {model_name}')\n    for (train_size, test_size, model_name_key), (precision, recall) in precision_recall_results.items():\n        if model_name == model_name_key:\n            plt.plot(recall, precision, label=f'Train: {train_size}, Test: {test_size}')\n    plt.xlabel('Recall')\n    plt.ylabel('Precision')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.show()\n\n\n# Create DataFrame for metrics results\nmetrics_df = pd.DataFrame(metrics_results, columns=['Train Size', 'Test Size', 'Model', 'Accuracy', 'Precision', 'Recall', 'F1-Score'])\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:33:47.562458Z","iopub.execute_input":"2024-04-15T14:33:47.562931Z","iopub.status.idle":"2024-04-15T14:34:55.373459Z","shell.execute_reply.started":"2024-04-15T14:33:47.562887Z","shell.execute_reply":"2024-04-15T14:34:55.372130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics_df","metadata":{"execution":{"iopub.status.busy":"2024-04-15T14:34:55.375891Z","iopub.execute_input":"2024-04-15T14:34:55.376280Z","iopub.status.idle":"2024-04-15T14:34:55.398301Z","shell.execute_reply.started":"2024-04-15T14:34:55.376245Z","shell.execute_reply":"2024-04-15T14:34:55.396734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}