{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\ndf = pd.read_csv('../input/amex-default-prediction/train_data.csv', nrows=120_000)\ndf.S_2 = pd.to_datetime(df.S_2)\ndf2 = pd.read_csv('../input/amex-default-prediction/train_labels.csv', nrows=120_000)\ndf = df.merge(df2,on='customer_ID',how='left')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-21T18:06:45.116118Z","iopub.execute_input":"2022-06-21T18:06:45.116801Z","iopub.status.idle":"2022-06-21T18:06:51.178708Z","shell.execute_reply.started":"2022-06-21T18:06:45.116756Z","shell.execute_reply":"2022-06-21T18:06:51.177515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:06:51.180785Z","iopub.execute_input":"2022-06-21T18:06:51.181185Z","iopub.status.idle":"2022-06-21T18:06:51.210935Z","shell.execute_reply.started":"2022-06-21T18:06:51.181151Z","shell.execute_reply":"2022-06-21T18:06:51.209656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1- Check target variable","metadata":{}},{"cell_type":"code","source":"df2.sample(3)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:06:51.212789Z","iopub.execute_input":"2022-06-21T18:06:51.213448Z","iopub.status.idle":"2022-06-21T18:06:51.226955Z","shell.execute_reply.started":"2022-06-21T18:06:51.213391Z","shell.execute_reply":"2022-06-21T18:06:51.226255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"customer_id is presented only, at least we don't see the time here, so the target is part of an state rather than a value of each step of the time series","metadata":{}},{"cell_type":"code","source":"# Target is between 0 and 1\ndf2.target.value_counts().plot(kind ='bar', title = 'barplot for target variable')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:06:51.228667Z","iopub.execute_input":"2022-06-21T18:06:51.229290Z","iopub.status.idle":"2022-06-21T18:06:51.401323Z","shell.execute_reply.started":"2022-06-21T18:06:51.229257Z","shell.execute_reply":"2022-06-21T18:06:51.400492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see imbalanced data where 0 is represented by people who didn't fall to default and 1 ioc, a bit higher the proportion than I expected, but looks ok for me. ","metadata":{}},{"cell_type":"markdown","source":"# X variables","metadata":{}},{"cell_type":"markdown","source":"The problem is a time series problem as we can see, we have a client, a datetime and several variables to study. If we select the first client in the dataframe and select a variable at random (in this case we select D_41 (related to delinquency), we can plot it as follow: ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize =(12,5))\nax.plot(df[df.customer_ID == df.customer_ID[0]].S_2, df[df.customer_ID == df.customer_ID[0]].D_41, color ='r')\n#ax.set_xlabel()\nax.set_title(f\"var D_41 from customer {df.customer_ID[0]}\")\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:06:51.402555Z","iopub.execute_input":"2022-06-21T18:06:51.403280Z","iopub.status.idle":"2022-06-21T18:06:51.820854Z","shell.execute_reply.started":"2022-06-21T18:06:51.403242Z","shell.execute_reply":"2022-06-21T18:06:51.819771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.a Delinquency Variables","metadata":{}},{"cell_type":"code","source":"#Select only delinquency variables \"D_\"\ndf_delinquency = df.filter(like='D_')","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:06:51.822829Z","iopub.execute_input":"2022-06-21T18:06:51.823319Z","iopub.status.idle":"2022-06-21T18:06:51.859363Z","shell.execute_reply.started":"2022-06-21T18:06:51.823273Z","shell.execute_reply":"2022-06-21T18:06:51.857958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_delinquency.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:06:51.861123Z","iopub.execute_input":"2022-06-21T18:06:51.861795Z","iopub.status.idle":"2022-06-21T18:06:52.616792Z","shell.execute_reply.started":"2022-06-21T18:06:51.861749Z","shell.execute_reply":"2022-06-21T18:06:52.615858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_missing_points(df):\n    fig, ax = plt.subplots(figsize = (20,5))\n    sns.heatmap(df.isnull(), cbar = False ,linecolor = 'w')\n    ax.set_title('Missing data points')\n    plt.show()\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:06:52.618305Z","iopub.execute_input":"2022-06-21T18:06:52.619347Z","iopub.status.idle":"2022-06-21T18:06:52.625085Z","shell.execute_reply.started":"2022-06-21T18:06:52.619305Z","shell.execute_reply":"2022-06-21T18:06:52.624109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_missing_points(df_delinquency)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:06:52.626652Z","iopub.execute_input":"2022-06-21T18:06:52.627081Z","iopub.status.idle":"2022-06-21T18:07:04.299683Z","shell.execute_reply.started":"2022-06-21T18:06:52.627047Z","shell.execute_reply":"2022-06-21T18:07:04.298724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check the distribution of each variable","metadata":{}},{"cell_type":"code","source":"ax = df_delinquency.hist(layout = (47, 2),\n                          alpha = 0.5, label ='x',\n                          figsize = (30,100), bins =20,)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:07:04.302569Z","iopub.execute_input":"2022-06-21T18:07:04.302929Z","iopub.status.idle":"2022-06-21T18:07:19.557144Z","shell.execute_reply.started":"2022-06-21T18:07:04.302896Z","shell.execute_reply":"2022-06-21T18:07:19.556062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_delinquency.D_68.unique()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:07:19.558461Z","iopub.execute_input":"2022-06-21T18:07:19.558899Z","iopub.status.idle":"2022-06-21T18:07:19.567056Z","shell.execute_reply.started":"2022-06-21T18:07:19.558868Z","shell.execute_reply":"2022-06-21T18:07:19.566130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In general, we can see the distribution in delinquency is far from normal distribution and you can see some variables are representing ordinal  (or similar) encoding, for instance D_117 or D_68. Another point of view is to consider the distribution with periods. Chris Deotte built an interesting visualizations https://www.kaggle.com/code/cdeotte/time-series-eda","metadata":{}},{"cell_type":"code","source":"def heatmap_corr(df):\n    corr = df.corr()\n\n    mask = np.triu(np.ones_like(corr, dtype = bool))\n    f, ax = plt.subplots(figsize=(30,12), nrows =1, ncols=2)\n    cmap = sns.diverging_palette(230,20, as_cmap = True)\n    sns.heatmap(corr, mask = mask, cmap=cmap, square = True, vmax =.3, \n               center =0, linewidths = .5, ax=ax[0])\n    ax[0].set_title('Correlation Heatmap')\n    \n\n    ax[1].hist(corr.unstack().values, bins =50)#.plot()\n    ax[1].set_title('Correlation Distribution')\n    plt.show()\n    \n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:07:19.568287Z","iopub.execute_input":"2022-06-21T18:07:19.568624Z","iopub.status.idle":"2022-06-21T18:07:19.577892Z","shell.execute_reply.started":"2022-06-21T18:07:19.568594Z","shell.execute_reply":"2022-06-21T18:07:19.577160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a matrix correlation\n\nheatmap_corr(df_delinquency)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:07:19.579115Z","iopub.execute_input":"2022-06-21T18:07:19.579539Z","iopub.status.idle":"2022-06-21T18:07:22.978981Z","shell.execute_reply.started":"2022-06-21T18:07:19.579510Z","shell.execute_reply":"2022-06-21T18:07:22.977995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In general the correlation is quite weak in most variables","metadata":{}},{"cell_type":"code","source":"def boxplot_variables(df):\n    #Boxplot for all D_ variables\n    fig, ax = plt.subplots(figsize=(21,10))\n    df.boxplot( rot=90)\n    #ax.set_xticklabels(rotation = 45)\n    ax.set_title('Boxplot Delinquency Variables')\n    plt.show()\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:07:22.995249Z","iopub.execute_input":"2022-06-21T18:07:22.995930Z","iopub.status.idle":"2022-06-21T18:07:23.000832Z","shell.execute_reply.started":"2022-06-21T18:07:22.995887Z","shell.execute_reply":"2022-06-21T18:07:23.000144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Boxplot for all D_ variables\nboxplot_variables(df_delinquency)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:07:23.002226Z","iopub.execute_input":"2022-06-21T18:07:23.002532Z","iopub.status.idle":"2022-06-21T18:07:39.624829Z","shell.execute_reply.started":"2022-06-21T18:07:23.002504Z","shell.execute_reply":"2022-06-21T18:07:39.624083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"few variables contain a much higher space like D_65 and D_71, then D_69 have highest variance","metadata":{}},{"cell_type":"markdown","source":"I will continue updating this notebook for EDA, expect this can help with some other ideas for this competition. ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}