{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"padding:20px;color:white;margin:0;font-size:175%;text-align:center;display:fill;border-radius:5px;background-color:crimson;overflow:hidden;font-weight:500\">American Express - Default Prediction</div>\n<div style=\"padding:20px;color:black;margin:0;font-size:175%;text-align:center;display:fill;border-radius:5px;overflow:hidden;font-weight:500\">Competition Notebook</div>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"padding:5px;color:black;margin:0;font-size:175%;display:fill;border-radius:5px;overflow:hidden;font-weight:500\">Objective of the competition</div>\n<div style=\"padding:0.1px;color:black;margin:0;font-size:125%;display:fill;border-radius:5px;overflow:hidden;font-weight:10\">The objective of this competition is to predict the probability that a customer does not pay back their credit card balance amount in the future based on their monthly customer profile.</div>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"padding:5px;color:black;margin:0;font-size:175%;display:fill;border-radius:5px;overflow:hidden;font-weight:500\">Data Description</div>\nThe objective of this competition is to predict the probability that a customer does not pay back their credit card balance amount in the future based on their monthly customer profile. The target binary variable is calculated by observing 18 months performance window after the latest credit card statement, and if the customer does not pay due amount in 120 days after their latest statement date it is considered a default event.\n\nThe dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:\n\nD_* = Delinquency variables\nS_* = Spend variables\nP_* = Payment variables\nB_* = Balance variables\nR_* = Risk variables\nwith the following features being categorical:\n\n['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nYour task is to predict, for each customer_ID, the probability of a future payment default (target = 1).\n\nNote that the negative class has been subsampled for this dataset at 5%, and thus receives a 20x weighting in the scoring metric.","metadata":{}},{"cell_type":"markdown","source":"<div style=\"padding:10px;color:black;margin:0;font-size:175%;text-align:center;display:fill;border-radius:5px;overflow:hidden;font-weight:500\">Importing Libraries and Reading data</div>","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.colors\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold \nfrom sklearn.metrics import roc_auc_score, roc_curve, auc\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nimport warnings, gc\nwarnings.filterwarnings(\"ignore\")\ninit_notebook_mode(connected=True)\n\ntemp=dict(layout=go.Layout(font=dict(family=\"Franklin Gothic\", size=12), \n                           height=500, width=1000))\n\ntrain = pd.read_feather('../input/amexfeather/train_data.ftr')\ntrain = train.groupby('customer_ID').tail(1).set_index('customer_ID')\nprint(\"The training data begining - {}.\".format(train['S_2'].max().strftime('%m-%d-%Y')))\nprint(\"The training data ending - {}.\".format(train['S_2'].min().strftime('%m-%d-%Y')))\nprint(\"Number of customers in the training set - {:,.0f}\".format(train.shape[0]))\nprint(\"Number of features in the training set -  {} features.\".format(train.shape[1]))\n\ntest = pd.read_feather('../input/amexfeather/test_data.ftr')\ntest = test.groupby('customer_ID').tail(1).set_index('customer_ID')\nprint(\"The test data begining - {}.\".format(test['S_2'].min().strftime('%m-%d-%Y')))\nprint(\"The test data ending - {}.\".format(test['S_2'].max().strftime('%m-%d-%Y')))\nprint(\"Number of customers in the test - {:,.0f}\".format(test.shape[0]))\nprint(\"Number of features in the test -  {} features.\".format(test.shape[1]))\n\ndel test['S_2']\ngc.collect()\n\ntitles=['Delinquency '+str(i).split('_')[1] if i.startswith('D') else 'Spend '+str(i).split('_')[1] \n        if i.startswith('S') else 'Payment '+str(i).split('_')[1]  if i.startswith('P') \n        else 'Balance '+str(i).split('_')[1] if i.startswith('B') else \n        'Risk '+str(i).split('_')[1] for i in train.columns[:-1]]\ncat_cols=['Balance 30', 'Balance 38', 'Delinquency 63', 'Delinquency 64', 'Delinquency 66', 'Delinquency 68',\n          'Delinquency 114', 'Delinquency 116', 'Delinquency 117', 'Delinquency 120', 'Delinquency 126', 'Target']\ntest.columns=titles[1:]\ntitles.append('Target')\ntrain.columns=titles","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:30:38.157440Z","iopub.execute_input":"2022-08-10T06:30:38.158210Z","iopub.status.idle":"2022-08-10T06:31:41.421396Z","shell.execute_reply.started":"2022-08-10T06:30:38.158106Z","shell.execute_reply":"2022-08-10T06:31:41.420000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"padding:10px;color:black;margin:0;font-size:175%;text-align:center;display:fill;border-radius:5px;overflow:hidden;font-weight:500\">Data Analysis</div>","metadata":{}},{"cell_type":"code","source":"target=train.Target.value_counts(normalize=True)\ntarget.rename(index={1:'Default',0:'Paid'},inplace=True)\npal, color=['blue','red'], ['blue','red']\nfig=go.Figure()\nfig.add_trace(go.Pie(labels=target.index, values=target*100, \n                     showlegend=True,sort=False, \n                     marker=dict(colors=color,line=dict(color=pal,width=2.5)),\n                     hovertemplate = \"%{label} Accounts: %{value:.2f}%<extra></extra>\"))\nfig.update_layout(template=temp, title='Target Distribution', \n                  legend=dict(traceorder='reversed',y=1.05,x=0),\n                  uniformtext_minsize=15, uniformtext_mode='hide',width=700)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-10T06:32:13.881134Z","iopub.execute_input":"2022-08-10T06:32:13.881546Z","iopub.status.idle":"2022-08-10T06:32:13.994580Z","shell.execute_reply.started":"2022-08-10T06:32:13.881512Z","shell.execute_reply":"2022-08-10T06:32:13.993689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target=pd.DataFrame(data={'Default':train.groupby('Spend 2')['Target'].mean()*100})\ntarget['Paid']=np.abs(train.groupby('Spend 2')['Target'].mean()-1)*100\nrgb=['rgba'+str(matplotlib.colors.to_rgba(i,0.7)) for i in pal]\nfig=go.Figure()\nfig.add_trace(go.Bar(x=target.index, y=target.Paid, name='Paid',\n                     text=target.Paid, texttemplate='%{text:.0f}%', \n                     textposition='inside',insidetextanchor=\"middle\",\n                     marker=dict(color=color[0],line=dict(color=pal[0],width=1.5)),\n                     hovertemplate = \"<b>%{x}</b><br>Paid accounts: %{y:.2f}%\"))\nfig.add_trace(go.Bar(x=target.index, y=target.Default, name='Default',\n                     text=target.Default, texttemplate='%{text:.0f}%', \n                     textposition='inside',insidetextanchor=\"middle\",\n                     marker=dict(color=color[1],line=dict(color=pal[1],width=1.5)),\n                     hovertemplate = \"<b>%{x}</b><br>Default accounts: %{y:.2f}%\"))\nfig.update_layout(template=temp,title='Distribution of Default by Day', \n                  barmode='relative', yaxis_ticksuffix='%', width=1400,\n                  legend=dict(orientation=\"h\", traceorder=\"reversed\", yanchor=\"bottom\",y=1.1,xanchor=\"left\", x=0))\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:32:24.181961Z","iopub.execute_input":"2022-08-10T06:32:24.182412Z","iopub.status.idle":"2022-08-10T06:32:24.291951Z","shell.execute_reply.started":"2022-08-10T06:32:24.182377Z","shell.execute_reply":"2022-08-10T06:32:24.290805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_df=train.reset_index().groupby('Spend 2')['customer_ID'].nunique().reset_index()\nfig=go.Figure()\nfig.add_trace(go.Scatter(x=plot_df['Spend 2'], \n                         y=plot_df['customer_ID'], mode='lines',\n                         line=dict(color=pal[0], width=3), \n                         hovertemplate = ''))\nfig.update_layout(template=temp, title=\"Frequency of Customer Statements\", \n                  hovermode=\"x unified\", width=800,height=500,\n                  xaxis_title='Statement Date', yaxis_title='Number of Statements Issued')\nfig.show()\ndel train['Spend 2']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:32:41.321292Z","iopub.execute_input":"2022-08-10T06:32:41.321739Z","iopub.status.idle":"2022-08-10T06:32:41.780700Z","shell.execute_reply.started":"2022-08-10T06:32:41.321699Z","shell.execute_reply":"2022-08-10T06:32:41.779462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols=[col for col in train.columns if (col.startswith(('D','T'))) & (col not in cat_cols[:-1])]\nplot_df=train[cols]\nfig, ax = plt.subplots(18,5, figsize=(16,54))\nfig.suptitle('Distribution of Delinquency Variables',fontsize=16)\nrow=0\ncol=[0,1,2,3,4]*18\nfor i, column in enumerate(plot_df.columns[:-1]):\n    if (i!=0)&(i%5==0):\n        row+=1\n    sns.kdeplot(x=column, hue='Target', palette=pal[::-1], hue_order=[1,0], \n                label=['Default','Paid'], data=plot_df, \n                fill=True, linewidth=2, legend=False, ax=ax[row,col[i]])\n    ax[row,col[i]].tick_params(left=False,bottom=False)\n    ax[row,col[i]].set(title='\\n\\n{}'.format(column), xlabel='', ylabel=('Density' if i%5==0 else ''))\nfor i in range(2,5):\n    ax[17,i].set_visible(False)\nhandles, _ = ax[0,0].get_legend_handles_labels() \nfig.legend(labels=['Default','Paid'], handles=reversed(handles), ncol=2, bbox_to_anchor=(0.18, 0.983))\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:32:50.841307Z","iopub.execute_input":"2022-08-10T06:32:50.841747Z","iopub.status.idle":"2022-08-10T06:35:44.808418Z","shell.execute_reply.started":"2022-08-10T06:32:50.841708Z","shell.execute_reply":"2022-08-10T06:35:44.806920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=plot_df.iloc[:,:-1].corr()\nmask=np.triu(np.ones_like(corr, dtype=bool))[1:,:-1]\ncorr=corr.iloc[1:,:-1].copy()\nfig, ax = plt.subplots(figsize=(48,48))   \nsns.heatmap(corr, mask=mask, vmin=-1, vmax=1, center=0, annot=True, fmt='.2f', \n            cmap='coolwarm', annot_kws={'fontsize':10,'fontweight':'bold'}, cbar=False)\nax.tick_params(left=False,bottom=False)\nax.set_xticklabels(ax.get_xticklabels(), rotation=45, horizontalalignment='right',fontsize=12)\nax.set_yticklabels(ax.get_yticklabels(), fontsize=12)\nplt.title('Correlations between Payment Variables\\n', fontsize=30)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:36:19.172826Z","iopub.execute_input":"2022-08-10T06:36:19.173932Z","iopub.status.idle":"2022-08-10T06:36:46.267199Z","shell.execute_reply.started":"2022-08-10T06:36:19.173885Z","shell.execute_reply":"2022-08-10T06:36:46.265759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,4, figsize=(16,5))\nfig.suptitle('Relationships between Delinquency Variables,\\nLog-Transformed',fontsize=16)\nax[0].hexbin(x='Delinquency 74', y='Delinquency 75', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[0].set(xlabel='Delinquency 74',ylabel='Delinquency 75')\nax[0].text(1, 4, 'Correlation: {:.2f}'.format(plot_df[['Delinquency 74','Delinquency 75']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[1].hexbin(x='Delinquency 58', y='Delinquency 74', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[1].set(xlabel='Delinquency 58',ylabel='Delinquency 74')\nax[1].text(0.3, 4.2, 'Correlation: {:.2f}'.format(plot_df[['Delinquency 58','Delinquency 74']].corr().iloc[1,0]),\n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[2].hexbin(x='Delinquency 113', y='Delinquency 115', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[2].set(xlabel='Delinquency 73',ylabel='Delinquency 137')\nax[2].text(2.15, 1.95, 'Correlation: {:.2f}'.format(plot_df[['Delinquency 113','Delinquency 115']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[3].hexbin(x='Delinquency 131', y='Delinquency 132', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[3].set(xlabel='Delinquency 131',ylabel='Delinquency 132')\nax[3].text(1.1, 5.9, 'Correlation: {:.2f}'.format(plot_df[['Delinquency 131','Delinquency 132']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nfor i in range(4):\n    ax[i].tick_params(left=False,bottom=False)\nsns.despine()\nplt.tight_layout(rect=[0, 0, 1, 0.99])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols=[col for col in train.columns if (col.startswith(('S','T'))) & (col not in cat_cols[:-1])]\nplot_df=train[cols]\nfig, ax = plt.subplots(5,5, figsize=(16,20))\nfig.suptitle('Distribution of Spend Variables',fontsize=16)\nrow=0\ncol=[0,1,2,3,4]*5\nfor i, column in enumerate(plot_df.columns[:-1]):\n    if (i!=0)&(i%5==0):\n        row+=1\n    sns.kdeplot(x=column, hue='Target', palette=pal[::-1], hue_order=[1,0], \n                label=['Default','Paid'], data=plot_df, \n                fill=True, linewidth=2, legend=False, ax=ax[row,col[i]])\n    ax[row,col[i]].tick_params(left=False,bottom=False)\n    ax[row,col[i]].set(title='\\n\\n{}'.format(column), xlabel='', ylabel=('Density' if i%5==0 else ''))\nfor i in range(1,5):\n    ax[4,i].set_visible(False)\nhandles, _ = ax[0,0].get_legend_handles_labels() \nfig.legend(labels=['Default','Paid'], handles=reversed(handles), ncol=2, bbox_to_anchor=(0.18, 0.985))\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=plot_df.corr()\nmask=np.triu(np.ones_like(corr, dtype=bool))[1:,:-1]\ncorr=corr.iloc[1:,:-1].copy()\nfig, ax = plt.subplots(figsize=(16,12))   \nsns.heatmap(corr, mask=mask, vmin=-1, vmax=1, center=0, annot=True, fmt='.2f', \n            cmap='coolwarm', annot_kws={'fontsize':10,'fontweight':'bold'}, cbar=False)\nax.tick_params(left=False,bottom=False)\nax.set_xticklabels(ax.get_xticklabels(), rotation=45, horizontalalignment='right',fontsize=12)\nax.set_yticklabels(ax.get_yticklabels(), fontsize=12)\nplt.title('Correlations between Spend Variables\\n', fontsize=16)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,4, figsize=(16,5))\nfig.suptitle('Relationships between Spend Variables,\\nLog-Transformed',fontsize=16)\nax[0].hexbin(x='Spend 24', y='Spend 22', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[0].set(xlabel='Spend 24',ylabel='Spend 22')\nax[0].text(-70, 4, 'Correlation: {:.2f}'.format(plot_df[['Spend 24','Spend 22']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[1].hexbin(x='Spend 7', y='Spend 3', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[1].set(xlabel='Spend 7',ylabel='Spend 3')\nax[1].text(0.4, 4.15, 'Correlation: {:.2f}'.format(plot_df[['Spend 7','Spend 3']].corr().iloc[1,0]),\n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[2].hexbin(x='Spend 15', y='Spend 8', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[2].set(xlabel='Spend 15',ylabel='Spend 8')\nax[2].text(1.2, 1.28, 'Correlation: {:.2f}'.format(plot_df[['Spend 15','Spend 8']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[3].hexbin(x='Spend 11', y='Spend 15', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[3].set(xlabel='Spend 11',ylabel='Spend 15')\nax[3].text(.5,5.5, 'Correlation: {:.2f}'.format(plot_df[['Spend 11','Spend 15']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nfor i in range(4):\n    ax[i].tick_params(left=False,bottom=False)\nsns.despine()\nplt.tight_layout(rect=[0, 0, 1, 0.99])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols=[col for col in train.columns if (col.startswith(('P','T'))) & (col not in cat_cols[:-1])]\nplot_df=train[cols]\nfig, ax = plt.subplots(1,3, figsize=(16,5))\nfig.suptitle('Distribution of Payment Variables',fontsize=16)\nfor i, col in enumerate(plot_df.columns[:-1]):\n    sns.kdeplot(x=col, hue='Target', palette=pal[::-1], hue_order=[1,0], \n                label=['Default','Paid'], data=plot_df, \n                fill=True, linewidth=2, legend=False, ax=ax[i])\n    ax[i].tick_params(left=False,bottom=False)\n    ax[i].set(title='{}'.format(col), xlabel='', ylabel=('Density' if i==0 else ''))\nhandles, _ = ax[0].get_legend_handles_labels() \nfig.legend(labels=['Default','Paid'], handles=reversed(handles), ncol=2, bbox_to_anchor=(0.18, 1))\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=plot_df.corr()\nmask=np.triu(np.ones_like(corr, dtype=bool))[1:,:-1]\ncorr=corr.iloc[1:,:-1].copy()\nfig, ax = plt.subplots(figsize=(7,5)) \nsns.heatmap(corr, mask=mask, vmin=-1, vmax=1, center=0, annot=True, fmt='.2f', \n            cmap='coolwarm', annot_kws={'fontsize':12,'fontweight':'bold'})\nax.tick_params(left=False,bottom=False)\nax.set_xticklabels(ax.get_xticklabels(), rotation=45, horizontalalignment='right',fontsize=12)\nax.set_yticklabels(ax.get_yticklabels(), fontsize=12)\nplt.title('Correlations between Payment Variables\\n', fontsize=16)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(16,5))\nfig.suptitle('Relationships between Payment Variables,\\nLog-Transformed',fontsize=16)\nax[0].hexbin(x='Payment 2', y='Payment 3', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[0].text(-.2,2.2, 'Correlation: {:.2f}'.format(plot_df[['Payment 2','Payment 3']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[0].set(xlabel='Payment 2',ylabel='Payment 3')\nax[1].hexbin(x='Payment 3', y='Payment 4', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[1].text(-.6,1.35, 'Correlation: {:.2f}'.format(plot_df[['Payment 3','Payment 4']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[1].set(xlabel='Payment 3',ylabel='Payment 4')\nax[2].hexbin(x='Payment 4', y='Payment 2', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[2].text(.25,1.1, 'Correlation: {:.2f}'.format(plot_df[['Payment 4','Payment 2']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[2].set(xlabel='Payment 4',ylabel='Payment 2')\nfor i in range(3):\n    ax[i].tick_params(left=False,bottom=False)\nsns.despine()\nplt.tight_layout(rect=[0, 0, 1, 0.99])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols=[col for col in train.columns if (col.startswith(('B','T'))) & (col not in cat_cols[:-1])]\nplot_df=train[cols]\nfig, ax = plt.subplots(8,5, figsize=(16,32))\nfig.suptitle('Distribution of Balance Variables',fontsize=16)\nrow=0\ncol=[0,1,2,3,4]*8\nfor i, column in enumerate(plot_df.columns[:-1]):\n    if (i!=0)&(i%5==0):\n        row+=1\n    sns.kdeplot(x=column, hue='Target', palette=pal[::-1], hue_order=[1,0], \n                label=['Default','Paid'], data=plot_df, \n                fill=True, linewidth=2, legend=False, ax=ax[row,col[i]])\n    ax[row,col[i]].tick_params(left=False,bottom=False)\n    ax[row,col[i]].set(title='\\n\\n{}'.format(column), xlabel='', ylabel=('Density' if i%5==0 else ''))\nfor i in range(3,5):\n    ax[7,i].set_visible(False)\nhandles, _ = ax[0,0].get_legend_handles_labels() \nfig.legend(labels=['Default','Paid'], handles=reversed(handles), ncol=2, bbox_to_anchor=(0.18, 0.984))\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=plot_df.corr()\nmask=np.triu(np.ones_like(corr, dtype=bool))[1:,:-1]\ncorr=corr.iloc[1:,:-1].copy()\nfig, ax = plt.subplots(figsize=(24,22))   \nsns.heatmap(corr, mask=mask, vmin=-1, vmax=1, center=0, annot=True, fmt='.2f', \n            cmap='coolwarm', annot_kws={'fontsize':12,'fontweight':'bold'}, cbar=False)\nax.tick_params(left=False,bottom=False)\nax.set_xticklabels(ax.get_xticklabels(), rotation=45, horizontalalignment='right',fontsize=12)\nax.set_yticklabels(ax.get_yticklabels(), fontsize=12)\nplt.title('Correlations between Balance Variables\\n', fontsize=16)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(16,5))\nfig.suptitle('Relationships between Balance Variables,\\nLog-Transformed',fontsize=16)\nax[0].hexbin(x='Balance 23', y='Balance 7', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[0].text(.23,1.42, 'Correlation: {:.2f}'.format(plot_df[['Balance 23','Balance 7']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[0].set(xlabel='Balance 23',ylabel='Balance 7')\nax[1].hexbin(x='Balance 3', y='Balance 11', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[1].text(.3,1.85, 'Correlation: {:.2f}'.format(plot_df[['Balance 3','Balance 11']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[1].set(xlabel='Balance 3',ylabel='Balance 11')\nax[2].hexbin(x='Balance 11', y='Balance 2', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[2].text(.3,1.07, 'Correlation: {:.2f}'.format(plot_df[['Balance 11','Balance 2']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[2].set(xlabel='Balance 11',ylabel='Balance 2')\nfor i in range(3):\n    ax[i].tick_params(left=False,bottom=False)\nsns.despine()\nplt.tight_layout(rect=[0, 0, 1, 0.99])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols=[col for col in train.columns if (col.startswith(('R','T'))) & (col not in cat_cols[:-1])]\nplot_df=train[cols]\nfig, ax = plt.subplots(6,5, figsize=(16,24))\nfig.suptitle('Distribution of Risk Variables',fontsize=16)\nrow=0\ncol=[0,1,2,3,4]*6\nfor i, column in enumerate(plot_df.columns[:-1]):\n    if (i!=0)&(i%5==0):\n        row+=1\n    sns.kdeplot(x=column, hue='Target', palette=pal[::-1], hue_order=[1,0], \n                label=['Default','Paid'], data=plot_df, \n                fill=True, linewidth=2, legend=False, ax=ax[row,col[i]])\n    ax[row,col[i]].tick_params(left=False,bottom=False)\n    ax[row,col[i]].set(title='\\n\\n{}'.format(column), xlabel='', ylabel=('Density' if i%5==0 else ''))\nfor i in range(3,5):\n    ax[5,i].set_visible(False)\nhandles, _ = ax[0,0].get_legend_handles_labels() \nfig.legend(labels=['Default','Paid'], handles=reversed(handles), ncol=2, bbox_to_anchor=(0.18, 0.984))\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=plot_df.corr()\nmask=np.triu(np.ones_like(corr, dtype=bool))[1:,:-1]\ncorr=corr.iloc[1:,:-1].copy()\nfig, ax = plt.subplots(figsize=(24,18))   \nsns.heatmap(corr, mask=mask, vmin=-1, vmax=1, center=0, annot=True, fmt='.2f', \n            cmap='coolwarm', annot_kws={'fontsize':12,'fontweight':'bold'}, cbar=False)\nax.tick_params(left=False,bottom=False)\nax.set_xticklabels(ax.get_xticklabels(), rotation=45, horizontalalignment='right',fontsize=12)\nax.set_yticklabels(ax.get_yticklabels(), fontsize=12)\nplt.title('Correlations between Risk Variables\\n', fontsize=16)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(16,5))\nfig.suptitle('Relationships between Risk Variables,\\nLog-Transformed',fontsize=16)\nax[0].hexbin(x='Risk 8', y='Risk 5', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[0].text(5,35.7, 'Correlation: {:.2f}'.format(plot_df[['Risk 8','Risk 5']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[0].set(xlabel='Risk 8',ylabel='Risk 5')\nax[1].hexbin(x='Risk 3', y='Risk 16', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[1].text(1.3,14.3, 'Correlation: {:.2f}'.format(plot_df[['Risk 3','Risk 16']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[1].set(xlabel='Risk 3',ylabel='Risk 16')\nax[2].hexbin(x='Risk 20', y='Risk 17', data=plot_df, bins='log', gridsize=40, cmap='coolwarm')\nax[2].text(7,1.02, 'Correlation: {:.2f}'.format(plot_df[['Risk 20','Risk 17']].corr().iloc[1,0]), \n           ha=\"center\", va=\"center\",bbox=dict(boxstyle=\"round,pad=0.3\",fc=\"white\"))\nax[2].set(xlabel='Risk 20',ylabel='Risk 17')\nfor i in range(3):\n    ax[i].tick_params(left=False,bottom=False)\nsns.despine()\nplt.tight_layout(rect=[0, 0, 1, 0.99])\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = make_subplots(rows=4, cols=3, \n                    subplot_titles=cat_cols[:-1], \n                    vertical_spacing=0.1)\nrow=0\nc=[1,2,3]*5\nplot_df=train[cat_cols]\nfor i,col in enumerate(cat_cols[:-1]):\n    if i%3==0:\n        row+=1\n    plot_df[col]=plot_df[col].astype(object)\n    df=plot_df.groupby(col)['Target'].value_counts().rename('count').reset_index().replace('',np.nan)\n    \n    fig.add_trace(go.Bar(x=df[df.Target==1][col], y=df[df.Target==1]['count'],\n                         marker_color=rgb[1], marker_line=dict(color=pal[1],width=2), \n                         hovertemplate='Value %{x} Frequency = %{y}',\n                         name='Default', showlegend=(True if i==0 else False)),\n                  row=row, col=c[i])\n    fig.add_trace(go.Bar(x=df[df.Target==0][col], y=df[df.Target==0]['count'],\n                         marker_color=rgb[0], marker_line=dict(color=pal[0],width=2),\n                         hovertemplate='Value %{x} Frequency = %{y}',\n                         name='Paid', showlegend=(True if i==0 else False)),\n                  row=row, col=c[i])\n    if i%3==0:\n        fig.update_yaxes(title='Frequency',row=row,col=c[i])\nfig.update_layout(template=temp,title=\"Distribution of Categorical Variables\",\n                  legend=dict(orientation=\"h\",yanchor=\"bottom\",y=1.03,xanchor=\"right\",x=0.2),\n                  barmode='group',height=1500,width=900)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=train.corr()\ncorr=corr['Target'].sort_values(ascending=False)[1:-1]\npal=sns.color_palette(\"Reds_r\",135).as_hex()\nrgb=['rgba'+str(matplotlib.colors.to_rgba(i,0.7)) for i in pal]\nfig = go.Figure()\nfig.add_trace(go.Bar(x=corr[corr>=0], y=corr[corr>=0].index, \n                     marker_color=rgb, orientation='h', \n                     marker_line=dict(color=pal,width=2), name='',\n                     hovertemplate='%{y} correlation with target: %{x:.3f}',\n                     showlegend=False))\npal=sns.color_palette(\"Blues\",100).as_hex()\nrgb=['rgba'+str(matplotlib.colors.to_rgba(i,0.7)) for i in pal]\nfig.add_trace(go.Bar(x=corr[corr<0], y=corr[corr<0].index, \n                     marker_color=rgb[25:], orientation='h', \n                     marker_line=dict(color=pal[25:],width=2), name='',\n                     hovertemplate='%{y} correlation with target: %{x:.3f}',\n                     showlegend=False))\nfig.update_layout(template=temp,title=\"Feature Correlations with Target\",\n                  xaxis_title=\"Correlation\", margin=dict(l=150),\n                  height=3000, width=700, hovermode='closest')\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null=round((train.isna().sum()/train.shape[0]*100),2).sort_values(ascending=False).astype(str)+('%')\nnull=null.to_frame().rename(columns={0:'Missing %'})\nnull.head(30)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"padding:10px;color:black;margin:0;font-size:175%;text-align:center;display:fill;border-radius:5px;overflow:hidden;font-weight:500\">Default Prediction</div>","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n\n\ndef plot_roc(y_val,y_prob):\n    colors=px.colors.qualitative.Prism\n    fig=go.Figure()\n    fig.add_trace(go.Scatter(x=np.linspace(0,1,11), y=np.linspace(0,1,11), \n                             name='Random Chance',mode='lines', showlegend=False,\n                             line=dict(color=\"Black\", width=1, dash=\"dot\")))\n    for i in range(len(y_val)):\n        y=y_val[i]\n        prob=y_prob[i]\n        fpr, tpr, _ = roc_curve(y, prob)\n        roc_auc = auc(fpr,tpr)\n        fig.add_trace(go.Scatter(x=fpr, y=tpr, line=dict(color=colors[::-1][i+1], width=3), \n                                 hovertemplate = 'True positive rate = %{y:.3f}<br>False positive rate = %{x:.3f}',\n                                 name='Fold {}:  Gini = {:.3f}, AUC = {:.3f}'.format(i+1, gini[i],roc_auc)))\n    fig.update_layout(template=temp, title=\"Cross-Validation ROC Curves\", \n                      hovermode=\"x unified\", width=700,height=600,\n                      xaxis_title='False Positive Rate (1 - Specificity)',\n                      yaxis_title='True Positive Rate (Sensitivity)',\n                      legend=dict(orientation='v', y=.07, x=1, xanchor=\"right\",\n                                  bordercolor=\"black\", borderwidth=.5))\n    fig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc = LabelEncoder()\nfor col in cat_cols[:-1]:\n    train[col] = enc.fit_transform(train[col])\n    test[col] = enc.transform(test[col])\n\nX=train.drop(['Target'],axis=1)\ny=train['Target']\ny_valid, gbm_val_probs, gbm_test_preds, gini=[],[],[],[]\nft_importance=pd.DataFrame(index=X.columns)\nsk_fold = StratifiedKFold(n_splits=10, shuffle=True, random_state=21)\nfor fold, (train_idx, val_idx) in enumerate(sk_fold.split(X, y)):\n    \n    print(\"\\nFold {}\".format(fold+1))\n    X_train, y_train = X.iloc[train_idx,:], y[train_idx]\n    X_val, y_val = X.iloc[val_idx,:], y[val_idx]\n    print(\"Train shape: {}, {}, Valid shape: {}, {}\\n\".format(\n        X_train.shape, y_train.shape, X_val.shape, y_val.shape))\n    \n    params = {'boosting_type': 'gbdt',\n              'n_estimators': 1000,\n              'num_leaves': 50,\n              'learning_rate': 0.05,\n              'colsample_bytree': 0.9,\n              'min_child_samples': 2000,\n              'max_bins': 500,\n              'reg_alpha': 2,\n              'objective': 'binary',\n              'random_state': 21}\n    \n    gbm = LGBMClassifier(**params).fit(X_train, y_train, \n                                       eval_set=[(X_train, y_train), (X_val, y_val)],\n                                       callbacks=[early_stopping(200), log_evaluation(500)],\n                                       eval_metric=['auc','binary_logloss'])\n    gbm_prob = gbm.predict_proba(X_val)[:,1]\n    gbm_val_probs.append(gbm_prob)\n    y_valid.append(y_val)\n    \n    y_pred=pd.DataFrame(data={'prediction':gbm_prob})\n    y_true=pd.DataFrame(data={'target':y_val.reset_index(drop=True)})\n    gini_score=amex_metric(y_true = y_true, y_pred = y_pred)\n    gini.append(gini_score)\n    \n    auc_score=roc_auc_score(y_val, gbm_prob)\n    gbm_test_preds.append(gbm.predict_proba(test)[:,1])    \n    ft_importance[\"Importance_Fold\"+str(fold)]=gbm.feature_importances_    \n    print(\"Validation Gini: {:.5f}, AUC: {:.4f}\".format(gini_score,auc_score))\n    \n    del X_train, y_train, X_val, y_val\n    _ = gc.collect()\n    \ndel X, y\nplot_roc(y_valid, gbm_val_probs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ft_importance['avg']=ft_importance.mean(axis=1)\nft_importance=ft_importance.avg.nlargest(50).sort_values(ascending=True)\n\npal=sns.color_palette(\"YlGnBu\", 65).as_hex()\nfig=go.Figure()\nfor i in range(len(ft_importance.index)):\n    fig.add_shape(dict(type=\"line\", y0=i, y1=i, x0=0, x1=ft_importance[i], \n                       line_color=pal[::-1][i],opacity=0.8,line_width=4))\nfig.add_trace(go.Scatter(x=ft_importance, y=ft_importance.index, mode='markers', \n                         marker_color=pal[::-1], marker_size=8,\n                         hovertemplate='%{y} Importance = %{x:.0f}<extra></extra>'))\nfig.update_layout(template=temp,title='LGBM Feature Importance<br>Top 50', \n                  margin=dict(l=150,t=80),\n                  xaxis=dict(title='Importance', zeroline=False),\n                  yaxis_showgrid=False, height=1000, width=800)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"padding:10px;color:black;margin:0;font-size:175%;text-align:center;display:fill;border-radius:5px;overflow:hidden;font-weight:500\">Submission</div>","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsub['prediction']=np.mean(gbm_test_preds, axis=0)\n\ndf=pd.DataFrame(data={'Target':sub['prediction'].apply(lambda x: 1 if x>0.5 else 0)})\ndf=df.Target.value_counts(normalize=True)\ndf.rename(index={1:'Default',0:'Paid'},inplace=True)\npal, color=['blue','red'], ['blue','red']\nfig=go.Figure()\nfig.add_trace(go.Pie(labels=df.index, values=df*100, \n                     showlegend=True,sort=False, \n                     marker=dict(colors=color,line=dict(color=pal,width=2.5)),\n                     hovertemplate = \"%{label} Accounts: %{value:.2f}%<extra></extra>\"))\nfig.update_layout(template=temp, title='Predicted Target Distribution', \n                  legend=dict(traceorder='reversed',y=1.05,x=0),\n                  uniformtext_minsize=15, uniformtext_mode='hide',width=700)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)\ndisplay(sub.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}