{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3727003,"sourceType":"datasetVersion","datasetId":2213609}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.colors\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.decomposition import PCA\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nfrom sklearn.model_selection import StratifiedKFold , TimeSeriesSplit\nfrom sklearn.metrics import classification_report, roc_auc_score,accuracy_score, roc_curve, auc\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nfrom itertools import cycle\nimport warnings, gc\nwarnings.filterwarnings('ignore', category=UserWarning, module='lightgbm')\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\")\ntemp=dict(layout=go.Layout(font=dict(family=\"Franklin Gothic\", size=12), \n                           height=500, width=1000))\n\n#Custom Color Palette 🎨\ncustom_colors = [\"#70d6ff\",\"#ff4d6d\",\"#8338ec\",\"#90cf8e\",\"#ffd670\"]\ncustomPalette = sns.set_palette(sns.color_palette(custom_colors))\nsns.palplot(sns.color_palette(custom_colors),size=1.2)\nplt.tick_params(axis='both', labelsize=0, length = 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:28:19.460234Z","iopub.execute_input":"2025-01-25T00:28:19.460758Z","iopub.status.idle":"2025-01-25T00:28:19.604015Z","shell.execute_reply.started":"2025-01-25T00:28:19.460714Z","shell.execute_reply":"2025-01-25T00:28:19.602848Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reading data and exploring","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_feather('../input/amexfeather/train_data.ftr')\ndf_test = pd.read_feather('../input/amexfeather/test_data.ftr')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:28:25.368566Z","iopub.execute_input":"2025-01-25T00:28:25.368957Z","iopub.status.idle":"2025-01-25T00:28:42.185686Z","shell.execute_reply.started":"2025-01-25T00:28:25.368917Z","shell.execute_reply":"2025-01-25T00:28:42.184316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_rows = df_train.shape[0]\nnum_features = df_train.shape[1] \n\nnumerical_cols = df_train.select_dtypes(include=[\"number\"]).columns\nnumerical_cols = numerical_cols.tolist()\nnumerical_cols.remove(\"target\")\ncategorical_cols = df_train.select_dtypes(include=['object', 'category']).columns\ncategorical_cols = categorical_cols.tolist()\ncategorical_cols.remove(\"customer_ID\")\ndatetime_cols = df_train.select_dtypes(include=['datetime64']).columns\n\n\nprint(f\"Number of rows: {num_rows}\")\nprint(f\"Number of features: {num_features}, including the customer_ID\")\nprint(f\"Number of numerical columns: {len(numerical_cols)}\")\nprint(f\"Number of date columns: {len(datetime_cols)}\")\nprint(f\"Number of categorical columns: {len(categorical_cols)}\")\n\nprint(f\"\\nNumerical columns: {numerical_cols}\")\nprint(f\"Date columns: {datetime_cols}\")\nprint(f\"Categorical columns: {categorical_cols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:28:52.514251Z","iopub.execute_input":"2025-01-25T00:28:52.514653Z","iopub.status.idle":"2025-01-25T00:28:53.361972Z","shell.execute_reply.started":"2025-01-25T00:28:52.514624Z","shell.execute_reply":"2025-01-25T00:28:53.360847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_rows = df_test.shape[0]\nnum_features = df_test.shape[1]\n\nnumerical_cols = df_test.select_dtypes(include=[\"number\"]).columns\nnumerical_cols = numerical_cols.tolist()\n\n\ncategorical_cols = df_test.select_dtypes(include=['object', 'category']).columns\ncategorical_cols = categorical_cols.tolist()\ncategorical_cols.remove(\"customer_ID\")\ndatetime_cols = df_train.select_dtypes(include=['datetime64']).columns\n\n\nprint(f\"Number of rows: {num_rows}\")\nprint(f\"Number of features: {num_features}, including the customer_ID\")\nprint(f\"Number of numerical columns: {len(numerical_cols)}\")\nprint(f\"Number of date columns: {len(datetime_cols)}\")\nprint(f\"Number of categorical columns: {len(categorical_cols)}\")\n\nprint(f\"\\nNumerical columns: {numerical_cols}\")\nprint(f\"Date columns: {datetime_cols}\")\nprint(f\"Categorical columns: {categorical_cols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:28:56.315837Z","iopub.execute_input":"2025-01-25T00:28:56.316382Z","iopub.status.idle":"2025-01-25T00:28:58.243781Z","shell.execute_reply.started":"2025-01-25T00:28:56.316333Z","shell.execute_reply":"2025-01-25T00:28:58.242235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['S_2'] = pd.to_datetime(df_train['S_2'])\ndf_test['S_2'] = pd.to_datetime(df_test['S_2'])\n\n# Find the earliest and latest dates\nstart_date_train = df_train['S_2'].min()\nend_date_train = df_train['S_2'].max()\nstart_date_test = df_test['S_2'].min()\nend_date_test = df_test['S_2'].max()\n\n\n# # Print the result\nprint(f\"The train data exists from {start_date_train.strftime('%Y-%m-%d')} to {end_date_train.strftime('%Y-%m-%d')}.\")\nprint(f\"The test data exists from {start_date_test.strftime('%Y-%m-%d')} to {end_date_test.strftime('%Y-%m-%d')}.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\nsns.histplot(data=df_train, x=\"S_2\", bins=100)\nplt.title(\"Distribution of statements by time for train data\", fontsize=16)\nplt.xlabel(\"count\", fontsize=14)\nplt.ylabel(\"n_records\", fontsize=14);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:29:04.856910Z","iopub.execute_input":"2025-01-25T00:29:04.857259Z","iopub.status.idle":"2025-01-25T00:29:07.088278Z","shell.execute_reply.started":"2025-01-25T00:29:04.857232Z","shell.execute_reply":"2025-01-25T00:29:07.087060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test['S_2'] = pd.to_datetime(df_test['S_2'])\nplt.figure(figsize=(16, 5))\nsns.histplot(data=df_test, x=\"S_2\", bins = 100)\nplt.title(\"Distribution of statements by time for test data\", fontsize=16)\nplt.xlabel(\"count\", fontsize=14)\nplt.ylabel(\"n_records\", fontsize=14);","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are multiple entries per customer, we need to pick only one unique row per customer to model","metadata":{}},{"cell_type":"code","source":"customer_presence = df_train.groupby(['customer_ID','target']).size().reset_index().rename(columns={0:'presence'})\ncustomer_presence[\"presence\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:33:28.960657Z","iopub.execute_input":"2025-01-24T23:33:28.961189Z","iopub.status.idle":"2025-01-24T23:33:30.595380Z","shell.execute_reply.started":"2025-01-24T23:33:28.961132Z","shell.execute_reply":"2025-01-24T23:33:30.594215Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Majority of the data is present in all 13 months but some customers are only repeated 1 to 12 times","metadata":{}},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5))\ntrain_sc = df_train.customer_ID.value_counts().value_counts().sort_index(ascending=False).rename('Train statements per customer')\nax1.pie(train_sc, labels=train_sc.index)\nax1.set_title(train_sc.name)\ntest_sc = df_test.customer_ID.value_counts().value_counts().sort_index(ascending=False).rename('Test statements per customer')\nax2.pie(test_sc, labels=test_sc.index)\nax2.set_title(test_sc.name)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:33:30.596604Z","iopub.execute_input":"2025-01-24T23:33:30.596921Z","iopub.status.idle":"2025-01-24T23:33:32.956850Z","shell.execute_reply.started":"2025-01-24T23:33:30.596892Z","shell.execute_reply":"2025-01-24T23:33:32.955371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"we can notice that how many rows/statements there are per customer. We see that 80 % of the customers have 13 statements; the other 20 % of the customers have between 1 and 12 statements. We need to have unqiue entry per customer to model hence the next step. \n\nI had two approaches to select the unique rows from train data set\n1) Reading only unique customer_ID rows for Train dataset that too the earliest defaulted row or latest row/statment per customer if not defaulted (commented it out for this notebook as it takes lot of time)\n2) selecting latest month statement per customer\n\ni ran the code on both iteratively and experimented ","metadata":{}},{"cell_type":"code","source":"# convert S_2 column into datetime and sort the dataframe by customer_ID and date (S_2); selecting latest statement\ndf_train['S_2'] = pd.to_datetime(df_train['S_2'])\ndf_train = df_train.sort_values(['customer_ID', 'S_2'])\ndf_train = df_train.sort_values('S_2').groupby('customer_ID').tail(1)\n\n# Group by customer_ID and apply custom logic of slecting earliest default row per customer or otherwise latest row/statement\n# def select_row(group):\n#     # Check if there's any row where target == 1\n#     first_target_1 = group[group['target'] == 1]\n#     if not first_target_1.empty:\n#         return first_target_1.iloc[0]  # Select the first row with target == 1\n#     else:\n#         return group.iloc[-1]  # Otherwise, select the last row in the group\n\n# df_train = df_train.groupby('customer_ID').apply(select_row).reset_index(drop=True)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:33:32.958074Z","iopub.execute_input":"2025-01-24T23:33:32.958412Z","iopub.status.idle":"2025-01-24T23:34:05.090056Z","shell.execute_reply.started":"2025-01-24T23:33:32.958382Z","shell.execute_reply":"2025-01-24T23:34:05.088927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Reading only unique customer_ID rows for Test data that too the latest row/statement by date since there are no target variables to pick different approaches","metadata":{}},{"cell_type":"code","source":"df_test = df_test.sort_values(['customer_ID', 'S_2'])\ndf_test['S_2'] = pd.to_datetime(df_test['S_2'])\ndf_test = df_test.sort_values('S_2').groupby('customer_ID').tail(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:34:05.091169Z","iopub.execute_input":"2025-01-24T23:34:05.091491Z","iopub.status.idle":"2025-01-24T23:35:14.774294Z","shell.execute_reply.started":"2025-01-24T23:34:05.091442Z","shell.execute_reply":"2025-01-24T23:35:14.772888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.replace([np.inf, -np.inf], np.nan, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:14.776779Z","iopub.execute_input":"2025-01-24T23:35:14.777544Z","iopub.status.idle":"2025-01-24T23:35:15.839179Z","shell.execute_reply.started":"2025-01-24T23:35:14.777436Z","shell.execute_reply":"2025-01-24T23:35:15.837602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['D_66'] = df_train['D_66'].fillna(0).astype('category') # converting nan values to 0 since there are only two value counts (1 and nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:15.840688Z","iopub.execute_input":"2025-01-24T23:35:15.841227Z","iopub.status.idle":"2025-01-24T23:35:15.849876Z","shell.execute_reply.started":"2025-01-24T23:35:15.841194Z","shell.execute_reply":"2025-01-24T23:35:15.848642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmp = df_train.isna().sum().div(len(df_train)).mul(100).sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:15.850704Z","iopub.execute_input":"2025-01-24T23:35:15.851052Z","iopub.status.idle":"2025-01-24T23:35:16.311880Z","shell.execute_reply.started":"2025-01-24T23:35:15.851022Z","shell.execute_reply":"2025-01-24T23:35:16.310273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"displaying the distribution of null values by percentage","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,1, figsize=(25,10))\nsns.barplot(x=tmp[:100].index, y=tmp[:100].values, ax=ax[0])\nsns.barplot(x=tmp[100:].index, y=tmp[100:].values, ax=ax[1])\nax[0].set_ylabel(\"Percentage [%]\"), ax[1].set_ylabel(\"Percentage [%]\")\nax[0].tick_params(axis='x', rotation=90); ax[1].tick_params(axis='x', rotation=90)\nplt.suptitle(\"Amount of missing data\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:16.313274Z","iopub.execute_input":"2025-01-24T23:35:16.313644Z","iopub.status.idle":"2025-01-24T23:35:18.814348Z","shell.execute_reply.started":"2025-01-24T23:35:16.313614Z","shell.execute_reply":"2025-01-24T23:35:18.812859Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Based on the graph dopping all the columns with more than 80% missing values","metadata":{}},{"cell_type":"code","source":"null_columns_to_drop = tmp[tmp>80].index.tolist()\ndf_train_adc = df_train.drop(null_columns_to_drop, axis =1)\ndf_test_adc = df_test.drop(null_columns_to_drop, axis =1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:18.815716Z","iopub.execute_input":"2025-01-24T23:35:18.816055Z","iopub.status.idle":"2025-01-24T23:35:19.608370Z","shell.execute_reply.started":"2025-01-24T23:35:18.816027Z","shell.execute_reply":"2025-01-24T23:35:19.607028Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA\n\nColumns in the dataset are divided by the organisers in the following groups:\n\nD_*: Delinquency variables\n\nS_*: Spend variables\n\nP_*: Payment variables\n\nB_*: Balance variables\n\nR_*: Risk variables\n\nList item\nFollowing features are categorical: B_30, B_38, D_63, D_64, D_66, D_68, D_114, D_116, D_117, D_120, D_126.\n\nS_2: contains a timestamp","metadata":{}},{"cell_type":"code","source":"D_columns = df_train_adc.filter(like='D_', axis=1).columns.tolist()\nB_columns = df_train_adc.filter(like='B_', axis=1).columns.tolist()\nS_columns = df_train_adc.filter(like='S_', axis=1).columns.tolist()\nP_columns = df_train_adc.filter(like='P_', axis=1).columns.tolist()\nR_columns = df_train_adc.filter(like='R_', axis=1).columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:19.609481Z","iopub.execute_input":"2025-01-24T23:35:19.609882Z","iopub.status.idle":"2025-01-24T23:35:19.717641Z","shell.execute_reply.started":"2025-01-24T23:35:19.609842Z","shell.execute_reply":"2025-01-24T23:35:19.716309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels=['Delinquency', 'Spend','Payment','Balance','Risk']\nvalues= [len(D_columns), len(S_columns),len(P_columns), len(B_columns),len(R_columns)]\nfig_1 = go.Figure()\nfig_1.add_trace(go.Pie(values = values,labels = labels,hole = 0.6, \n                     hoverinfo ='label+percent'))\nfig_1.update_traces(textfont_size = 12, hoverinfo ='label+percent',textinfo ='label', \n                  showlegend = False,marker = dict(colors =[\"#70d6ff\",\"#ff9770\"]),\n                  title = dict(text = 'Feature Distribution'))  \nfig_1.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:19.719139Z","iopub.execute_input":"2025-01-24T23:35:19.719611Z","iopub.status.idle":"2025-01-24T23:35:20.062828Z","shell.execute_reply.started":"2025-01-24T23:35:19.719573Z","shell.execute_reply":"2025-01-24T23:35:20.061745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_class = pd.DataFrame({'count': df_train_adc.target.value_counts(),\n                             'percentage': df_train_adc['target'].value_counts() / df_train_adc.shape[0] * 100\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:20.064139Z","iopub.execute_input":"2025-01-24T23:35:20.064566Z","iopub.status.idle":"2025-01-24T23:35:20.079047Z","shell.execute_reply.started":"2025-01-24T23:35:20.064532Z","shell.execute_reply":"2025-01-24T23:35:20.077562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import plotly.graph_objects as go\nfig = go.Figure()\nfig.add_trace(go.Pie(values = target_class['count'],labels = target_class.index,hole = 0.6, \n                     hoverinfo ='label+percent'))\nfig.update_traces(textfont_size = 12, hoverinfo ='label+percent',textinfo ='label', \n                  showlegend = False,marker = dict(colors =[\"#90cf8e\",\"#ff70a6\"]),\n                  title = dict(text = 'Target Distribution'))  \nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:20.080422Z","iopub.execute_input":"2025-01-24T23:35:20.080950Z","iopub.status.idle":"2025-01-24T23:35:20.120904Z","shell.execute_reply.started":"2025-01-24T23:35:20.080903Z","shell.execute_reply":"2025-01-24T23:35:20.118737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stat_plot = df_train_adc.reset_index().groupby('S_2')['customer_ID'].nunique().reset_index()\nfig = go.Figure()\nfig.add_trace(go.Scatter(x = stat_plot['S_2'], y = stat_plot['customer_ID']))\nfig.update_layout(title=\"Customer Statements\", width = 800, height = 600,xaxis_title ='Statement Date',\n                  paper_bgcolor='rgb(0,0,0,0)',plot_bgcolor='rgb(0,0,0,0)') \nfig['data'][0]['line']['color']=\"#ff9770\"\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:20.129872Z","iopub.execute_input":"2025-01-24T23:35:20.130326Z","iopub.status.idle":"2025-01-24T23:35:20.628937Z","shell.execute_reply.started":"2025-01-24T23:35:20.130292Z","shell.execute_reply":"2025-01-24T23:35:20.627698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del_cols = [c for c in df_train_adc.columns if (c.startswith(('D','t'))) & (c not in categorical_cols)]\ndf_del = df_train_adc[del_cols]\nspd_cols = [c for c in df_train_adc.columns if (c.startswith(('S','t'))) & (c not in categorical_cols)]\ndf_spd = df_train_adc[spd_cols]\npay_cols = [c for c in df_train_adc.columns if (c.startswith(('P','t'))) & (c not in categorical_cols)]\ndf_pay = df_train_adc[pay_cols]\nbal_cols = [c for c in df_train_adc.columns if (c.startswith(('B','t'))) & (c not in categorical_cols)]\ndf_bal = df_train_adc[bal_cols]\nris_cols = [c for c in df_train_adc.columns if (c.startswith(('R','t'))) & (c not in categorical_cols)]\ndf_ris = df_train_adc[ris_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:20.631802Z","iopub.execute_input":"2025-01-24T23:35:20.632121Z","iopub.status.idle":"2025-01-24T23:35:20.740501Z","shell.execute_reply.started":"2025-01-24T23:35:20.632095Z","shell.execute_reply":"2025-01-24T23:35:20.739287Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA of delinquency variables","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(29, 3, figsize = (35,150))\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(del_cols) - 1:\n        sns.kdeplot(x = del_cols[i], hue='target', data = df_del, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nfig.suptitle('Distribution of Delinquency Variables', fontsize = 35, x = 0.5, y = 1)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:35:20.741597Z","iopub.execute_input":"2025-01-24T23:35:20.741950Z","iopub.status.idle":"2025-01-24T23:37:18.899443Z","shell.execute_reply.started":"2025-01-24T23:35:20.741919Z","shell.execute_reply":"2025-01-24T23:37:18.896975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize =(11,11))\ncorr = df_del.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths =.6, cmap = custom_colors)\nplt.title('Correlation of Delinquency Variables')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:18.900981Z","iopub.execute_input":"2025-01-24T23:37:18.901315Z","iopub.status.idle":"2025-01-24T23:37:26.689606Z","shell.execute_reply.started":"2025-01-24T23:37:18.901285Z","shell.execute_reply":"2025-01-24T23:37:26.688406Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"checking for high correlation to remove columns for avoiding mulitcollinearity","metadata":{}},{"cell_type":"code","source":"high_pos_corr = corr[corr > 0.9]\n\n# Stack the correlation matrix and reset the index\nstacked_corr = (\n    high_pos_corr.stack()\n    .reset_index()\n    .rename(columns={0: 'correlation'})\n)\n\n# Ensure unique pairs by keeping only where level_0 < level_1\nstacked_corr = stacked_corr[stacked_corr['level_0'] < stacked_corr['level_1']]\n\n# Sort by correlation value\nstacked_corr = stacked_corr.sort_values(by='correlation', ascending=False)\n\nprint(stacked_corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:26.690979Z","iopub.execute_input":"2025-01-24T23:37:26.691382Z","iopub.status.idle":"2025-01-24T23:37:26.725269Z","shell.execute_reply.started":"2025-01-24T23:37:26.691345Z","shell.execute_reply":"2025-01-24T23:37:26.724070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# high_pos_corr = corr[corr < -0.9]\n\n# # Stack the correlation matrix and reset the index\n# stacked_corr = (\n#     high_pos_corr.stack()\n#     .reset_index()\n#     .rename(columns={0: 'correlation'})\n# )\n\n# # Ensure unique pairs by keeping only where level_0 < level_1\n# stacked_corr = stacked_corr[stacked_corr['level_0'] < stacked_corr['level_1']]\n\n# # Sort by correlation value\n# stacked_corr = stacked_corr.sort_values(by='correlation', ascending=False)\n\n# print(stacked_corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:26.726541Z","iopub.execute_input":"2025-01-24T23:37:26.726841Z","iopub.status.idle":"2025-01-24T23:37:26.731511Z","shell.execute_reply.started":"2025-01-24T23:37:26.726816Z","shell.execute_reply":"2025-01-24T23:37:26.730332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#columns to drop to avoid multicollinearity\nD_drop_columns = [\"D_62\", \"D_143\", \"D_141\", \"D_103\", \"D_118\", \"D_74\", \"D_58\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:26.732664Z","iopub.execute_input":"2025-01-24T23:37:26.733083Z","iopub.status.idle":"2025-01-24T23:37:26.758346Z","shell.execute_reply.started":"2025-01-24T23:37:26.733044Z","shell.execute_reply":"2025-01-24T23:37:26.756927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA of Spend variables","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(8, 3, figsize = (16,18))\nfig.suptitle('Distribution of Spend Variables', fontsize = 15, x = 0.5, y = 1)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(spd_cols) - 1:\n        sns.kdeplot(x = spd_cols[i], hue ='target', data = df_spd, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:26.759851Z","iopub.execute_input":"2025-01-24T23:37:26.760604Z","iopub.status.idle":"2025-01-24T23:37:53.957399Z","shell.execute_reply.started":"2025-01-24T23:37:26.760553Z","shell.execute_reply":"2025-01-24T23:37:53.956001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (11,11))\ncorr = df_spd.corr()\nmask = np.triu(np.ones_like(corr, dtype=bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths = .6, cmap = custom_colors)\nplt.title('Correlation of Spend Variables')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:53.959155Z","iopub.execute_input":"2025-01-24T23:37:53.959556Z","iopub.status.idle":"2025-01-24T23:37:55.357091Z","shell.execute_reply.started":"2025-01-24T23:37:53.959519Z","shell.execute_reply":"2025-01-24T23:37:55.355653Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA of payment variables","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize = (12,4))\nfig.suptitle('Distribution of Payment Variables',fontsize = 15)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(pay_cols) - 1:\n        sns.kdeplot(x = pay_cols[i], hue ='target', data = df_pay, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:55.358625Z","iopub.execute_input":"2025-01-24T23:37:55.359078Z","iopub.status.idle":"2025-01-24T23:37:59.301733Z","shell.execute_reply.started":"2025-01-24T23:37:55.359044Z","shell.execute_reply":"2025-01-24T23:37:59.300518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (6,6))\ncorr = df_pay.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust = True, center = 0,square = True, linewidths = .6, cmap = custom_colors)\nplt.title('Correlation of Payment Variables')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:59.303237Z","iopub.execute_input":"2025-01-24T23:37:59.303698Z","iopub.status.idle":"2025-01-24T23:37:59.587043Z","shell.execute_reply.started":"2025-01-24T23:37:59.303655Z","shell.execute_reply":"2025-01-24T23:37:59.585925Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA of Balance variables","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 4, figsize = (15,24))\nfig.suptitle('Distribution of Balance Variables',fontsize = 15, x = 0.5, y = 1)\nfor i, ax in enumerate(axes.reshape(-1)):\n    if i < len(bal_cols) - 1:\n        sns.kdeplot(x = bal_cols[i], hue ='target', data = df_bal, fill = True, ax = ax, palette =[\"#e63946\",\"#8338ec\"])\n        ax.tick_params()\n        ax.xaxis.get_label()\n        ax.set_ylabel('')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:37:59.588359Z","iopub.execute_input":"2025-01-24T23:37:59.588705Z","iopub.status.idle":"2025-01-24T23:38:53.431394Z","shell.execute_reply.started":"2025-01-24T23:37:59.588659Z","shell.execute_reply":"2025-01-24T23:38:53.430084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (11,11))\ncorr = df_bal.corr()\nmask = np.triu(np.ones_like(corr, dtype = bool))\nsns.heatmap(corr, mask = mask, robust=True, center = 0,square = True, linewidths =.6, cmap = custom_colors)\nplt.title('Correlation of Balance Variables')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:38:53.433032Z","iopub.execute_input":"2025-01-24T23:38:53.433583Z","iopub.status.idle":"2025-01-24T23:38:56.016254Z","shell.execute_reply.started":"2025-01-24T23:38:53.433524Z","shell.execute_reply":"2025-01-24T23:38:56.015068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"high_pos_corr = corr[corr > 0.9]\n\n# Stack the correlation matrix and reset the index\nstacked_corr = (\n    high_pos_corr.stack()\n    .reset_index()\n    .rename(columns={0: 'correlation'})\n)\n\n# Ensure unique pairs by keeping only where level_0 < level_1\nstacked_corr = stacked_corr[stacked_corr['level_0'] < stacked_corr['level_1']]\n\n# Sort by correlation value\nstacked_corr = stacked_corr.sort_values(by='correlation', ascending=False)\n\nprint(stacked_corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:38:56.017474Z","iopub.execute_input":"2025-01-24T23:38:56.017802Z","iopub.status.idle":"2025-01-24T23:38:56.036375Z","shell.execute_reply.started":"2025-01-24T23:38:56.017767Z","shell.execute_reply":"2025-01-24T23:38:56.035172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"B_drop_columns = [\"B_11\", \"B_13\", 'B_23', 'B_1' ,'B_2', \"B_14\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:38:56.037953Z","iopub.execute_input":"2025-01-24T23:38:56.038383Z","iopub.status.idle":"2025-01-24T23:38:56.054817Z","shell.execute_reply.started":"2025-01-24T23:38:56.038351Z","shell.execute_reply":"2025-01-24T23:38:56.053413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_adc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:38:56.056216Z","iopub.execute_input":"2025-01-24T23:38:56.056658Z","iopub.status.idle":"2025-01-24T23:38:56.249240Z","shell.execute_reply.started":"2025-01-24T23:38:56.056616Z","shell.execute_reply":"2025-01-24T23:38:56.248039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA of Categorical Variables","metadata":{}},{"cell_type":"code","source":"fig = make_subplots(rows=4, cols=3, \n                    subplot_titles=categorical_cols[:-1], \n                    vertical_spacing=0.1)\npal=['#016CC9','#DEB078']\nrow=0\nc=[1,2,3]*5\nplot_df= df_train_adc[['D_63', 'D_64', 'D_68', 'B_30','D_66', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'target']]\nfor i,col in enumerate(categorical_cols[:-1]):\n    if i%3==0:\n        row+=1\n    plot_df[col]=plot_df[col].astype(object)\n    df=plot_df.groupby(col)['target'].value_counts().rename('count').reset_index().replace('',np.nan)\n    \n    fig.add_trace(go.Bar(x=df[df.target==1][col], y=df[df.target==1]['count'],\n                          marker_line=dict(color=pal[1],width=2), \n                         hovertemplate='Value %{x} Frequency = %{y}',\n                         name='Default', showlegend=(True if i==0 else False)),\n                  row=row, col=c[i])\n    fig.add_trace(go.Bar(x=df[df.target==0][col], y=df[df.target==0]['count'],\n                          marker_line=dict(color=pal[0],width=2),\n                         hovertemplate='Value %{x} Frequency = %{y}',\n                         name='Paid', showlegend=(True if i==0 else False)),\n                  row=row, col=c[i])\n    if i%3==0:\n        fig.update_yaxes(title='Frequency',row=row,col=c[i])\nfig.update_layout(template=temp,title=\"Distribution of Categorical Variables\",\n                  legend=dict(orientation=\"h\",yanchor=\"bottom\",y=1.03,xanchor=\"right\",x=0.2),\n                  barmode='group',height=1500,width=900)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:38:56.250672Z","iopub.execute_input":"2025-01-24T23:38:56.251107Z","iopub.status.idle":"2025-01-24T23:38:57.319517Z","shell.execute_reply.started":"2025-01-24T23:38:56.251066Z","shell.execute_reply":"2025-01-24T23:38:57.318266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"palette = cycle([\"#ffd670\",\"#70d6ff\",\"#ff4d6d\",\"#8338ec\",\"#90cf8e\"])\ntarg = df_train_adc.drop(columns = [\"customer_ID\", 'D_63','D_64']).corrwith(df_train_adc['target'], axis=0)\nval = [str(round(v ,1) *100) + '%' for v in targ.values]\nfig = go.Figure()\nfig.add_trace(go.Bar(y=targ.index, x= targ.values, orientation='h',text = val, marker_color = next(palette)))\nfig.update_layout(title = \"Correlation of variables with Target\",width = 750, height = 3500,\n                  paper_bgcolor='rgb(0,0,0,0)',plot_bgcolor='rgb(0,0,0,0)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:38:57.320629Z","iopub.execute_input":"2025-01-24T23:38:57.320939Z","iopub.status.idle":"2025-01-24T23:39:00.166401Z","shell.execute_reply.started":"2025-01-24T23:38:57.320913Z","shell.execute_reply":"2025-01-24T23:39:00.165289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmp_corr = df_train_adc.drop(columns = [\"customer_ID\", 'D_63','D_64']).corrwith(df_train_adc['target'], axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:00.167577Z","iopub.execute_input":"2025-01-24T23:39:00.167900Z","iopub.status.idle":"2025-01-24T23:39:03.075876Z","shell.execute_reply.started":"2025-01-24T23:39:00.167874Z","shell.execute_reply":"2025-01-24T23:39:03.074564Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Features with more than 50% correlation with target variable","metadata":{}},{"cell_type":"code","source":"indexes = tmp_corr[abs(tmp_corr) > 0.5].index\nindexes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:03.077141Z","iopub.execute_input":"2025-01-24T23:39:03.077544Z","iopub.status.idle":"2025-01-24T23:39:03.084851Z","shell.execute_reply.started":"2025-01-24T23:39:03.077501Z","shell.execute_reply":"2025-01-24T23:39:03.083795Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dropping highly correlated columns","metadata":{}},{"cell_type":"code","source":"df_train_adc = df_train_adc.drop(D_drop_columns, axis =1)\ndf_test_adc = df_test_adc.drop(D_drop_columns, axis =1)\n\ndf_train_adc = df_train_adc.drop(B_drop_columns, axis =1)\ndf_test_adc = df_test_adc.drop(B_drop_columns, axis =1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:03.086014Z","iopub.execute_input":"2025-01-24T23:39:03.086280Z","iopub.status.idle":"2025-01-24T23:39:05.084039Z","shell.execute_reply.started":"2025-01-24T23:39:03.086258Z","shell.execute_reply":"2025-01-24T23:39:05.082554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_adc.set_index(\"customer_ID\", inplace=True)\ndf_test_adc.set_index(\"customer_ID\", inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:05.085352Z","iopub.execute_input":"2025-01-24T23:39:05.085786Z","iopub.status.idle":"2025-01-24T23:39:05.094966Z","shell.execute_reply.started":"2025-01-24T23:39:05.085752Z","shell.execute_reply":"2025-01-24T23:39:05.093000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_cols = df_train_adc.select_dtypes(include=[\"number\"]).columns\nnumerical_cols = numerical_cols.tolist()\nnumerical_cols.remove(\"target\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:05.096399Z","iopub.execute_input":"2025-01-24T23:39:05.096760Z","iopub.status.idle":"2025-01-24T23:39:05.267810Z","shell.execute_reply.started":"2025-01-24T23:39:05.096727Z","shell.execute_reply":"2025-01-24T23:39:05.266341Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Imputing missing values ","metadata":{}},{"cell_type":"code","source":"num_imputer = SimpleImputer(strategy='mean') #can also use median to better handle outliers\ncat_imputer = SimpleImputer(strategy='most_frequent')\n\ndf_train_adc[numerical_cols] = num_imputer.fit_transform(df_train_adc[numerical_cols])\ndf_train_adc[categorical_cols] = cat_imputer.fit_transform(df_train_adc[categorical_cols])\n\ndf_test_adc[numerical_cols] = num_imputer.transform(df_test_adc[numerical_cols])\ndf_test_adc[categorical_cols] = cat_imputer.transform(df_test_adc[categorical_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:05.269247Z","iopub.execute_input":"2025-01-24T23:39:05.269647Z","iopub.status.idle":"2025-01-24T23:39:13.628534Z","shell.execute_reply.started":"2025-01-24T23:39:05.269609Z","shell.execute_reply":"2025-01-24T23:39:13.627329Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Label Encoding of categorical variables","metadata":{}},{"cell_type":"code","source":"label_encoders = {} \nfor col in categorical_cols:\n    le = LabelEncoder()\n    df_train_adc[col] = le.fit_transform(df_train_adc[col])\n    label_encoders[col] = le\nfor col in categorical_cols:\n    le = label_encoders[col]  # Retrieve the trained LabelEncoder\n    df_test_adc[col] = le.transform(df_test_adc[col])  # Transform the test column","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:13.629960Z","iopub.execute_input":"2025-01-24T23:39:13.630307Z","iopub.status.idle":"2025-01-24T23:39:17.278339Z","shell.execute_reply.started":"2025-01-24T23:39:13.630280Z","shell.execute_reply":"2025-01-24T23:39:17.277077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_adc.reset_index(inplace = True)\ndf_test_adc.reset_index(inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:17.279575Z","iopub.execute_input":"2025-01-24T23:39:17.279898Z","iopub.status.idle":"2025-01-24T23:39:17.363672Z","shell.execute_reply.started":"2025-01-24T23:39:17.279872Z","shell.execute_reply":"2025-01-24T23:39:17.362344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = df_train_adc.drop(columns=['customer_ID','target',\"S_2\"])\ntest_data = df_test_adc.drop(columns=['customer_ID','S_2'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:17.364952Z","iopub.execute_input":"2025-01-24T23:39:17.365367Z","iopub.status.idle":"2025-01-24T23:39:18.365856Z","shell.execute_reply.started":"2025-01-24T23:39:17.365335Z","shell.execute_reply":"2025-01-24T23:39:18.364705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Handling outliers","metadata":{}},{"cell_type":"code","source":"# Function to cap/floor outliers using IQR\ndef cap_outliers_iqr(df):\n    for col in df.select_dtypes(include=['number']).columns:\n        Q1 = df[col].quantile(0.25)  # First quartile\n        Q3 = df[col].quantile(0.75)  # Third quartile\n        IQR = Q3 - Q1                # Interquartile range\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n        df[col] = df[col].clip(lower=lower_bound, upper=upper_bound)  # Capping outliers\n    return df\n\n# Apply the function\ntrain_data_cleaned = cap_outliers_iqr(train_data)\nprint(train_data_cleaned)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:18.366924Z","iopub.execute_input":"2025-01-24T23:39:18.367234Z","iopub.status.idle":"2025-01-24T23:39:26.836167Z","shell.execute_reply.started":"2025-01-24T23:39:18.367208Z","shell.execute_reply":"2025-01-24T23:39:26.835152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"# Step 1: standardize the data (importanct for PCA)\nscaler = StandardScaler()\ndata_scaled = scaler.fit_transform(train_data)\n\n# Step 2: perform default PCA without n_components parameter\npca = PCA()\npca.fit(data_scaled)\n\n# Step 3: Compute cumulative explained variance\ncumulative_variance = np.cumsum(pca.explained_variance_ratio_)\n\n# Step 4: Plot the elbow curve\nplt.figure(figsize=(8, 5))\nplt.plot(range(1, len(cumulative_variance) + 1), cumulative_variance, marker='o', linestyle='--')\nplt.xlabel('Number of Components')\nplt.ylabel('Cumulative Explained Variance')\nplt.title('Explained Variance by PCA Components')\nplt.axhline(y=0.9, color='r', linestyle='--', label='90% Variance Threshold')  # Optional\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:26.836984Z","iopub.execute_input":"2025-01-24T23:39:26.837292Z","iopub.status.idle":"2025-01-24T23:39:33.524356Z","shell.execute_reply.started":"2025-01-24T23:39:26.837267Z","shell.execute_reply":"2025-01-24T23:39:33.522908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimal_components = np.argmax(cumulative_variance >= 0.9) + 1\nprint(f\"Optimal number of components: {optimal_components}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:33.525725Z","iopub.execute_input":"2025-01-24T23:39:33.526156Z","iopub.status.idle":"2025-01-24T23:39:33.531804Z","shell.execute_reply.started":"2025-01-24T23:39:33.526127Z","shell.execute_reply":"2025-01-24T23:39:33.530702Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"By selecting optimal n_components in PCA, we effectively reduced the dimensionality of the training dataset while retaining 90% of the variance present in the original data. This dimensionality reduction helps simplify the dataset, making it more computationally efficient to process, without losing significant information critical to the analysis.","metadata":{}},{"cell_type":"code","source":"pca = PCA(n_components=optimal_components)\npca_result_train = pca.fit_transform(train_data)\npca_result_test = pca.transform(test_data)\n\ndf_pca_train = pd.DataFrame(pca_result_train)\ndf_pca_test = pd.DataFrame(pca_result_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:33.532900Z","iopub.execute_input":"2025-01-24T23:39:33.533284Z","iopub.status.idle":"2025-01-24T23:39:50.033569Z","shell.execute_reply.started":"2025-01-24T23:39:33.533255Z","shell.execute_reply":"2025-01-24T23:39:50.032588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['target']  = df_train_adc['target'].values\ndf_pca_train['target'] = df_train_adc['target'].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:50.034686Z","iopub.execute_input":"2025-01-24T23:39:50.035021Z","iopub.status.idle":"2025-01-24T23:39:50.043643Z","shell.execute_reply.started":"2025-01-24T23:39:50.034993Z","shell.execute_reply":"2025-01-24T23:39:50.042356Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now we have two pairs of dataframes \n1) (train_data & test_data)\n2) (df_pca_train & df_pca_test) #reduced dimensions but also lost feature names/meaning  hence cannot be used for feature importance as it is the limitation of PCA\n\nwe can train models on either of these datasets (i have experimented on both, the benefit i saw is faster computation when PCA data was used but lower accuracy/amex gini metrics)\n\nin this notebook i am choosing traditional pair (train_data & test_data)","metadata":{}},{"cell_type":"code","source":"# Prepare features and target\nX = train_data.drop('target', axis=1)\ny = train_data['target']\n\nprint('Feature count:', len(X.columns))\nprint('X shape:', X.shape)\nprint('y shape:', y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:50.044790Z","iopub.execute_input":"2025-01-24T23:39:50.045077Z","iopub.status.idle":"2025-01-24T23:39:50.709170Z","shell.execute_reply.started":"2025-01-24T23:39:50.045055Z","shell.execute_reply":"2025-01-24T23:39:50.708149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n\n\ndef plot_roc(y_val,y_prob):\n    colors=px.colors.qualitative.Prism\n    fig=go.Figure()\n    fig.add_trace(go.Scatter(x=np.linspace(0,1,11), y=np.linspace(0,1,11),\n                             name='Random Chance',mode='lines', showlegend=False,\n                             line=dict(color=\"Black\", width=1, dash=\"dot\")))\n    for i in range(len(y_val)):\n        y=y_val[i]\n        prob=y_prob[i]\n        fpr, tpr, _ = roc_curve(y, prob)\n        roc_auc = auc(fpr,tpr)\n        fig.add_trace(go.Scatter(x=fpr, y=tpr, line=dict(color=colors[::-1][i+1], width=3),\n                                 hovertemplate = 'True positive rate = %{y:.3f}<br>False positive rate = %{x:.3f}',\n                                 name='Fold {}:  Gini = {:.3f}, AUC = {:.3f}'.format(i+1, gini[i],roc_auc)))\n    fig.update_layout( title=\"Cross-Validation ROC Curves\",\n                      hovermode=\"x unified\", width=700,height=600,\n                      xaxis_title='False Positive Rate (1 - Specificity)',\n                      yaxis_title='True Positive Rate (Sensitivity)',\n                      legend=dict(orientation='v', y=.07, x=1, xanchor=\"right\",\n                                  bordercolor=\"black\", borderwidth=.5))\n    fig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:50.710431Z","iopub.execute_input":"2025-01-24T23:39:50.710850Z","iopub.status.idle":"2025-01-24T23:39:50.725218Z","shell.execute_reply.started":"2025-01-24T23:39:50.710812Z","shell.execute_reply":"2025-01-24T23:39:50.723608Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As my next step i wanted to run GridSearchCV to find optimal parameters for LightGBMClassifier model but it is computaionally very expensive and takes veryyyyy long to get to best params. hence i chose parmeters from one of the submitted kaggle codes. \n\nLightGBM was chosen for this project due to its efficiency in handling large datasets, its fast training speed, and its ability to model complex relationships through gradient boosting. Additionally, its built-in support for handling missing values and optimizing for both memory usage and accuracy makes it ideal for high-dimensional numerical data like this.","metadata":{}},{"cell_type":"markdown","source":"# Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"# # Define LightGBM model\n# model = lgb.LGBMClassifier(\n#     objective='binary',\n#     early_stopping=50,\n#     verbose=-1\n# )\n\n# # Parameter grid\n# param_grid = {\n#     'n_estimators': [250,500,1000],\n#     'learning_rate': [0.01, 0.05, 0.1],\n#     'max_depth': [3, 5, 7],\n#     'num_leaves': [15, 31, 63],\n#     'min_child_samples': [1000, 2000, 500]\n# }\n\n# # Scoring metric\n# scorer = make_scorer(accuracy_score)\n\n# # Define GridSearchCV with PredefinedSplit\n# grid_search = GridSearchCV(\n#     estimator=model,\n#     param_grid=param_grid,\n#     scoring=scorer,\n#     cv=predefined_split,\n#     verbose=1,\n#     n_jobs=-1\n# )\n\n# # Fit the model\n# grid_search.fit(X_train_full, y_train_full)\n\n# # Results\n# print(\"Best Parameters:\", grid_search.best_params_)\n# print(\"Best Score:\", grid_search.best_score_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:50.726649Z","iopub.execute_input":"2025-01-24T23:39:50.727086Z","iopub.status.idle":"2025-01-24T23:39:50.751665Z","shell.execute_reply.started":"2025-01-24T23:39:50.727041Z","shell.execute_reply":"2025-01-24T23:39:50.750437Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"To train the model, I considered two approaches: a TimeSeriesSplit for preserving the temporal structure of the data, given its temporal nature, and StratifiedKFold to address the high imbalance of the target variable.","metadata":{}},{"cell_type":"markdown","source":"# TimeSeriesSplit\n","metadata":{}},{"cell_type":"code","source":"tscv = TimeSeriesSplit(n_splits=5)\n\nparams = {'boosting_type': 'gbdt',\n              'n_estimators': 1000,\n              'num_leaves': 50,\n              'learning_rate': 0.05,\n              'colsample_bytree': 0.9,\n              'min_child_samples': 2000,\n              'max_bins': 500,\n              'reg_alpha': 2,\n              'objective': 'binary',\n              \"early_stopping_rounds\": 200,\n              'verbose': -1, \n              'random_state': 21}\n\naccuracy_scores = []\n\n# Iterate through Time Series splits\nfor split, (train_index, test_index) in enumerate(tscv.split(X)):\n    # Split data into train and test sets\n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n    print(f\"Split {split + 1}: Train size = {len(X_train)}, Test size = {len(X_test)}\")\n\n    clf = LGBMClassifier(**params).fit(X_train, y_train,\n                                       eval_set=[(X_train, y_train), (X_test, y_test)],\n                                                                             eval_metric=['auc','binary_logloss'])\n    # Make predictions on the test set\n    y_pred = clf.predict(X_test)\n\n    # Evaluate accuracy\n    accuracy = accuracy_score(y_test, y_pred)\n    accuracy_scores.append(accuracy)\n\n    y_pred_prob = clf.predict_proba(X_test)[:,1]\n    y_pred=pd.DataFrame(data={'prediction': y_pred_prob})\n    y_true_pred = pd.DataFrame(data={'target':y_test.values})\n    gini_score=amex_metric(y_true = y_true_pred, y_pred = y_pred)\n    print(f\"Split {split + 1}: Accuracy = {accuracy:.2f}, Gini = {gini_score:.2f}\")\n\n\n# Print overall results\nprint(\"\\nAccuracy scores for each split:\", accuracy_scores)\nprint(\"Mean Accuracy:\", np.mean(accuracy_scores))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:39:50.752779Z","iopub.execute_input":"2025-01-24T23:39:50.753075Z","iopub.status.idle":"2025-01-24T23:49:56.561394Z","shell.execute_reply.started":"2025-01-24T23:39:50.753044Z","shell.execute_reply":"2025-01-24T23:49:56.560019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission = pd.DataFrame()\n# submission['customer_ID'] = df_test_adc[\"customer_ID\"]\n# submission[\"prediction\"] = clf.predict_proba(test_data)[:,1]\n# submission.to_csv('submission_tsplit.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:49:56.563167Z","iopub.execute_input":"2025-01-24T23:49:56.563663Z","iopub.status.idle":"2025-01-24T23:50:27.468765Z","shell.execute_reply.started":"2025-01-24T23:49:56.563632Z","shell.execute_reply":"2025-01-24T23:50:27.467701Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# StratifiedKFold ","metadata":{}},{"cell_type":"code","source":"y_valid, gbm_val_probs, gbm_test_preds, gini=[],[],[],[]\nft_importance=pd.DataFrame(index=X.columns)\n\nsk_fold = StratifiedKFold(n_splits=5, shuffle=True, random_state=21)\n\nfor fold, (train_idx, val_idx) in enumerate(sk_fold.split(X, y)):\n\n    print(\"\\nFold {}\".format(fold+1))\n    X_train, y_train = X.iloc[train_idx,:], y[train_idx]\n    X_val, y_val = X.iloc[val_idx,:], y[val_idx]\n    print(\"Train shape: {}, {}, Valid shape: {}, {}\\n\".format(\n        X_train.shape, y_train.shape, X_val.shape, y_val.shape))\n\n    params = {'boosting_type': 'gbdt',\n              'n_estimators': 1000,\n              'num_leaves': 50,\n              'learning_rate': 0.05,\n              'colsample_bytree': 0.9,\n              'min_child_samples': 2000,\n              'max_bins': 500,\n              'reg_alpha': 2,\n              'objective': 'binary',\n              \"early_stopping_rounds\": 200,\n              'verbose': -1,\n              'random_state': 21}\n\n    gbm = LGBMClassifier(**params).fit(X_train, y_train,\n                                       eval_set=[(X_train, y_train), (X_val, y_val)],\n                                                                             eval_metric=['auc','binary_logloss'])\n    gbm_prob = gbm.predict_proba(X_val)[:,1]\n    gbm_val_probs.append(gbm_prob)\n    y_valid.append(y_val)\n\n    y_pred=pd.DataFrame(data={'prediction':gbm_prob})\n    y_true=pd.DataFrame(data={'target':y_val.reset_index(drop=True)})\n    gini_score=amex_metric(y_true = y_true, y_pred = y_pred)\n    gini.append(gini_score)\n\n    auc_score=roc_auc_score(y_val, gbm_prob)\n    gbm_test_preds.append(gbm.predict_proba(test_data)[:,1])\n    ft_importance[\"Importance_Fold\"+str(fold)]=gbm.feature_importances_\n    print(\"Validation Gini: {:.5f}, AUC: {:.4f}\".format(gini_score,auc_score))\n\n    del X_train, y_train, X_val, y_val\n    _ = gc.collect()\n\ndel X, y\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T23:50:27.469991Z","iopub.execute_input":"2025-01-24T23:50:27.470389Z","iopub.status.idle":"2025-01-25T00:09:32.851042Z","shell.execute_reply.started":"2025-01-24T23:50:27.470356Z","shell.execute_reply":"2025-01-25T00:09:32.849616Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['customer_ID'] = df_test_adc[\"customer_ID\"]\nsubmission[\"prediction\"] = gbm.predict_proba(test_data)[:,1]\nsubmission.to_csv('submission_skfold.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:09:32.852293Z","iopub.execute_input":"2025-01-25T00:09:32.852637Z","iopub.status.idle":"2025-01-25T00:10:12.999787Z","shell.execute_reply.started":"2025-01-25T00:09:32.852609Z","shell.execute_reply":"2025-01-25T00:10:12.998371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_roc(y_valid, gbm_val_probs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:13.001192Z","iopub.execute_input":"2025-01-25T00:10:13.001591Z","iopub.status.idle":"2025-01-25T00:10:13.182713Z","shell.execute_reply.started":"2025-01-25T00:10:13.001559Z","shell.execute_reply":"2025-01-25T00:10:13.181273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Importance","metadata":{}},{"cell_type":"code","source":"ft_importance['avg'] = ft_importance.mean(axis=1)\nft_importance = ft_importance.avg.nlargest(50).sort_values(ascending=True)\n\npal=sns.color_palette(\"YlGnBu\", 65).as_hex()\nfig=go.Figure()\nfor i in range(len(ft_importance.index)):\n    fig.add_shape(dict(type=\"line\", y0=i, y1=i, x0=0, x1=ft_importance[i],\n                       line_color=pal[::-1][i],opacity=0.8,line_width=4))\nfig.add_trace(go.Scatter(x=ft_importance, y=ft_importance.index, mode='markers',\n                         marker_color=pal[::-1], marker_size=8,\n                         hovertemplate='%{y} Importance = %{x:.0f}<extra></extra>'))\nfig.update_layout(template=temp,title='LGBM Feature Importance<br>Top 50',\n                  margin=dict(l=150,t=80),\n                  xaxis=dict(title='Importance', zeroline=False),\n                  yaxis_showgrid=False, height=1000, width=800)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:13.184427Z","iopub.execute_input":"2025-01-25T00:10:13.184979Z","iopub.status.idle":"2025-01-25T00:10:13.715083Z","shell.execute_reply.started":"2025-01-25T00:10:13.184925Z","shell.execute_reply":"2025-01-25T00:10:13.714045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"correlation_matrix = train_data[ft_importance.index.tolist()].corr()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:13.716174Z","iopub.execute_input":"2025-01-25T00:10:13.716510Z","iopub.status.idle":"2025-01-25T00:10:17.246280Z","shell.execute_reply.started":"2025-01-25T00:10:13.716451Z","shell.execute_reply":"2025-01-25T00:10:17.245136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (11,11))\nmask = np.triu(np.ones_like(correlation_matrix, dtype = bool))\nsns.heatmap(correlation_matrix, mask = mask, robust=True, center = 0,square = True, linewidths =.6, cmap = custom_colors)\nplt.title('Correlation of top 50 features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:17.247617Z","iopub.execute_input":"2025-01-25T00:10:17.247937Z","iopub.status.idle":"2025-01-25T00:10:18.023273Z","shell.execute_reply.started":"2025-01-25T00:10:17.247909Z","shell.execute_reply":"2025-01-25T00:10:18.022123Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SHAPLEY and LIME for Interpretation\nwe can notice how variables are influcing the prediction outcome based on colour coding of Shapley and LIME","metadata":{}},{"cell_type":"code","source":"import shap\nshap.initjs()\nfrom lime.lime_tabular import LimeTabularExplainer\nimport lime\n# Assuming the following variables are defined:\n# model: trained machine learning model (e.g., LightGBM)\n# X_train: training data (features)\n# X_test: test data (features)\n# y_test: test data (target)\n\n# -------------------------------\n# SHAP (SHapley Additive exPlanations)\n# -------------------------------\n\n# 1. Initialize the SHAP explainer\nexplainer_shap = shap.TreeExplainer(gbm)  # For tree-based models like LightGBM/XGBoost\n\n# 2. Compute SHAP values (using a subset of the data for performance)\nX_sample = test_data.sample(100)  # Use a smaller subset for SHAP computation\nshap_values = explainer_shap.shap_values(X_sample)\n\n# 3. Global feature importance\nshap.summary_plot(shap_values, X_sample)\n\n# 4. Individual prediction explanation\ninstance_index = 0  # Specify the row index to explain\nshap.force_plot(\n    explainer_shap.expected_value[1],  # For binary classification (index 1 = class 1)\n    shap_values[1][instance_index],   # SHAP values for the selected instance\n    test_data.iloc[instance_index]\n)\n\n# 5. SHAP dependence plot for a specific feature\nshap.dependence_plot('P_2', shap_values[1], X_sample)  # Replace 'P_2' with the desired feature\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:18.024819Z","iopub.execute_input":"2025-01-25T00:10:18.025311Z","iopub.status.idle":"2025-01-25T00:10:29.051854Z","shell.execute_reply.started":"2025-01-25T00:10:18.025267Z","shell.execute_reply":"2025-01-25T00:10:29.050594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shap.plots.force(explainer_shap.expected_value[0], shap_values[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:29.053123Z","iopub.execute_input":"2025-01-25T00:10:29.054005Z","iopub.status.idle":"2025-01-25T00:10:29.169657Z","shell.execute_reply.started":"2025-01-25T00:10:29.053967Z","shell.execute_reply":"2025-01-25T00:10:29.168403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shap.force_plot(explainer_shap.expected_value[0], shap_values[0][73,:], X_test.iloc[0,:], link=\"logit\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:29.177198Z","iopub.execute_input":"2025-01-25T00:10:29.177626Z","iopub.status.idle":"2025-01-25T00:10:29.187235Z","shell.execute_reply.started":"2025-01-25T00:10:29.177588Z","shell.execute_reply":"2025-01-25T00:10:29.186088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shap.force_plot(explainer_shap.expected_value[0], shap_values[0][42,:], X_test.iloc[0,:], link=\"logit\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:29.188577Z","iopub.execute_input":"2025-01-25T00:10:29.188934Z","iopub.status.idle":"2025-01-25T00:10:29.204235Z","shell.execute_reply.started":"2025-01-25T00:10:29.188905Z","shell.execute_reply":"2025-01-25T00:10:29.202876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shap.force_plot(explainer_shap.expected_value[0], shap_values[0], test_data, link=\"logit\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:29.205564Z","iopub.execute_input":"2025-01-25T00:10:29.206090Z","iopub.status.idle":"2025-01-25T00:10:30.275019Z","shell.execute_reply.started":"2025-01-25T00:10:29.206044Z","shell.execute_reply":"2025-01-25T00:10:30.273700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------------\n# LIME (Local Interpretable Model-agnostic Explanations)\n# -------------------------------\n\n# 1. Initialize the LIME explainer\nexplainer_lime = LimeTabularExplainer(\n    training_data=train_data.drop(columns=[\"target\"]).values,  # Training data (convert to numpy array)\n    mode='classification',         # 'classification' for classification tasks\n    feature_names=train_data.columns.tolist(), # Feature names\n    class_names=['target'],  \n    discretize_continuous=True ,   # Discretize continuous variables for interpretability\n    verbose=True, \n)\n\n# 2. Explain a single prediction\ninstance_index = 42  # Specify the row index to explain\ninstance = test_data.iloc[instance_index]\n\nexp = explainer_lime.explain_instance(\n    data_row=instance,\n    predict_fn=gbm.predict_proba,\n    num_features = 10\n)\n\nexp.show_in_notebook(show_table=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:30.276154Z","iopub.execute_input":"2025-01-25T00:10:30.276528Z","iopub.status.idle":"2025-01-25T00:10:56.880498Z","shell.execute_reply.started":"2025-01-25T00:10:30.276490Z","shell.execute_reply":"2025-01-25T00:10:56.879182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"instance_index = 73  # Specify the row index to explain\ninstance = test_data.iloc[instance_index]\n\nexp = explainer_lime.explain_instance(\n    data_row=instance,\n    predict_fn=gbm.predict_proba,\n    num_features=10\n)\n\nexp.show_in_notebook(show_table=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:56.881606Z","iopub.execute_input":"2025-01-25T00:10:56.881963Z","iopub.status.idle":"2025-01-25T00:10:57.536369Z","shell.execute_reply.started":"2025-01-25T00:10:56.881933Z","shell.execute_reply":"2025-01-25T00:10:57.535199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"instance_index = 42  # Specify the row index to explain\ninstance = test_data.iloc[instance_index]\n\nexp = explainer_lime.explain_instance(\n    data_row=instance,\n    predict_fn=gbm.predict_proba,\n    num_features=10\n)\n\nexp.show_in_notebook(show_table=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:57.537605Z","iopub.execute_input":"2025-01-25T00:10:57.537935Z","iopub.status.idle":"2025-01-25T00:10:58.195603Z","shell.execute_reply.started":"2025-01-25T00:10:57.537906Z","shell.execute_reply":"2025-01-25T00:10:58.194438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"instance_index = 2  # Specify the row index to explain\ninstance = test_data.iloc[instance_index]\n\nexp = explainer_lime.explain_instance(\n    data_row=instance,\n    predict_fn=gbm.predict_proba,\n    num_features=10\n)\n\nexp.show_in_notebook(show_table=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-25T00:10:58.196659Z","iopub.execute_input":"2025-01-25T00:10:58.196969Z","iopub.status.idle":"2025-01-25T00:10:58.849161Z","shell.execute_reply.started":"2025-01-25T00:10:58.196943Z","shell.execute_reply":"2025-01-25T00:10:58.847851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}