{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Importing Libararies\nimport pandas as pd\nimport numpy as np\nimport math\nimport gc\n\nimport plotly.graph_objs as go\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\n\nimport matplotlib.pyplot as plt\nimport matplotlib.ticker as mtick\nfrom matplotlib.ticker import FixedLocator\nimport matplotlib.gridspec as gridspec\n\nfrom termcolor import colored\nimport seaborn as sns\n\n# Define colors\npastel_pink = '#ffb6c1'\npastel_green = '#98fb98'\npastel_purple = '#c9a0dc'\nbeige_color = '#f5f5dc'","metadata":{"execution":{"iopub.status.busy":"2023-06-02T09:05:09.047061Z","iopub.execute_input":"2023-06-02T09:05:09.047524Z","iopub.status.idle":"2023-06-02T09:05:13.819786Z","shell.execute_reply.started":"2023-06-02T09:05:09.047481Z","shell.execute_reply":"2023-06-02T09:05:13.818491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Reading The Data**\nIn this EDA, we will be using feather format of the data by [munum](https://www.kaggle.com/datasets/munumbutt/amexfeather) to reduce the memory, also make it eaiser and faster to deal with data. ","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_feather(\"/kaggle/input/amexfeather/train_data.ftr\")\ntest_data = pd.read_feather(\"/kaggle/input/amexfeather/test_data.ftr\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:26:11.108636Z","iopub.execute_input":"2023-05-22T09:26:11.109124Z","iopub.status.idle":"2023-05-22T09:27:14.467526Z","shell.execute_reply.started":"2023-05-22T09:26:11.109086Z","shell.execute_reply":"2023-05-22T09:27:14.465641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **General Information about Data**","metadata":{}},{"cell_type":"code","source":"train_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:31:03.025996Z","iopub.execute_input":"2023-05-22T09:31:03.027431Z","iopub.status.idle":"2023-05-22T09:31:03.099816Z","shell.execute_reply.started":"2023-05-22T09:31:03.027366Z","shell.execute_reply":"2023-05-22T09:31:03.098501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntest_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:31:06.523865Z","iopub.execute_input":"2023-05-22T09:31:06.524646Z","iopub.status.idle":"2023-05-22T09:31:06.801440Z","shell.execute_reply.started":"2023-05-22T09:31:06.524600Z","shell.execute_reply":"2023-05-22T09:31:06.799938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for duplicates in customer ID and get the shape of the dataset\nprint(colored('Customer ID duplicates and dataset shape:', 'magenta'))\nprint(colored('Number of duplicated values in Customer ID: ', 'green'), len(train_data[train_data.customer_ID.duplicated(keep=False)]))\nprint(colored('Number of rows in the train labels dataset: ', 'green'), len(train_data), '\\n')\n\n# Check the distribution of the target variable\nprint(colored('Distribution of the target variable:', 'magenta'))\nlabel_stats = pd.DataFrame({'absolute': train_data.target.value_counts(),\n                            'relative': train_data.target.value_counts() / len(train_data)})\nlabel_stats['absolute upsampled'] =  label_stats.absolute * np.array([20, 1])\nlabel_stats['relative upsampled'] = label_stats['absolute upsampled'] / label_stats['absolute upsampled'].sum()\n\nlabel_stats_styled = label_stats.style.set_table_styles(\n    [{'selector': 'th',\n      'props': [('background-color', 'lightpink'), ('color', 'white'), ('border', '1px solid white'), ('text-align', 'center')]},\n     {'selector': 'td',\n      'props': [('background-color', 'lightgreen'), ('border', '1px solid white')]},\n     {'selector': 'tr:nth-of-type(even)',\n      'props': [('background-color', 'white')]},\n     {'selector': 'tr:nth-of-type(odd)',\n      'props': [('background-color', '#f2f2f2')]},\n     {'selector': 'caption',\n      'props': [('caption-side', 'top'), ('color', 'white'), ('background-color', 'magenta'), ('font-weight', 'bold')]}])\n\ndisplay(label_stats_styled)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:31:11.644474Z","iopub.execute_input":"2023-05-22T09:31:11.646401Z","iopub.status.idle":"2023-05-22T09:31:19.951882Z","shell.execute_reply.started":"2023-05-22T09:31:11.646342Z","shell.execute_reply":"2023-05-22T09:31:19.950307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntarget = train_data.target.value_counts(normalize=True)\ntarget.rename(index={1: 'Default', 0: 'Paid'}, inplace=True)\n\ncolors = [pastel_green, pastel_pink]\n\nfig = go.Figure()\nfig.add_trace(go.Pie(\n    labels=target.index, \n    values=target * 100,\n    showlegend=True,\n    marker=dict(colors=colors),\n    hovertemplate=\"%{label} Accounts: <b>%{value:.2f}</b>%<extra></extra>\"\n))\n\nfig.update_layout(\n    title={\n        \"text\":'<b>Default VS Paid Transaction</b><BR />Unbalanced dataset',\n        \"x\":0.035,\n        \"font_size\": 20,\n    },\n    uniformtext_minsize=15,\n    margin={'t':150, 'l':5}\n)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:31:38.360666Z","iopub.execute_input":"2023-05-22T09:31:38.361155Z","iopub.status.idle":"2023-05-22T09:31:38.800758Z","shell.execute_reply.started":"2023-05-22T09:31:38.361111Z","shell.execute_reply":"2023-05-22T09:31:38.799454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the ***customer_ID***'s are duplicated *5526331* times in dataset, because we see the statements of the customers' 18-month period, so In this case, we have some customers that are not duplicated, amount of *5120*. Let's keep in mind that data was subsampled to balance the class inbalance, so that these data rows are worth an investigation.","metadata":{}},{"cell_type":"markdown","source":"# ***Investigation Of Time related features***","metadata":{}},{"cell_type":"code","source":"gc.collect()\n# Extract year and month from datetime column for train and test dataframes\nfor df in [train_data, test_data]:\n    df['year'] = df['S_2'].dt.year\n    df['month'] = df['S_2'].dt.month","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:31:44.670259Z","iopub.execute_input":"2023-05-22T09:31:44.670757Z","iopub.status.idle":"2023-05-22T09:31:48.089450Z","shell.execute_reply.started":"2023-05-22T09:31:44.670714Z","shell.execute_reply":"2023-05-22T09:31:48.087883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Show the distribution of statements over years for train/test daa\ngc.collect()\nfig = make_subplots(rows=1, cols=2)\n\ndata = train_data['year'].value_counts()\nfig.add_trace(go.Bar(x=data.index, y=data.values, name='train', marker_color='#F5B7B1'),\n             row=1, col=1)\n\ndata = test_data['year'].value_counts()\nfig.add_trace(go.Bar(x=data.index, y=data.values, name='test', marker_color='#ABEBC6'),\n             row=1, col=2)\n\nfig.update_layout(template='ggplot2', \n                  title={\n                      \"text\": \"<b>Train/Test Year Distrubution</b> <BR />Train-2018 / Test--2019<br> <br> \",\n                      \"x\":0.035,\n                      \"font_size\": 20,\n                  },\n                  paper_bgcolor='beige',\n                  plot_bgcolor='beige'\n) ","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:31:51.615359Z","iopub.execute_input":"2023-05-22T09:31:51.616586Z","iopub.status.idle":"2023-05-22T09:31:53.623702Z","shell.execute_reply.started":"2023-05-22T09:31:51.616529Z","shell.execute_reply":"2023-05-22T09:31:53.622250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nfig = make_subplots(rows=1, cols=2)\ndata = train_data.reset_index().groupby(['year', 'month'])['customer_ID'].count()\ndata.index = data.index.to_series().apply(lambda x: f'{x[0]}-{x[1]}').values\nfig.add_trace(go.Bar(x=data.index, y=data.values, name='train', marker_color='#F5B7B1'), row=1, col=1)\n\n\ndata = test_data.reset_index().groupby(['year', 'month'])['customer_ID'].count()\ndata.index = data.index.to_series().apply(lambda x: f'{x[0]}-{x[1]}').values\nfig.add_trace(go.Bar(x=data.index, y=data.values, name='test', marker_color='#82E0AA'), row=1, col=2)\n\n\nfig.update_layout(template='ggplot2',\n                    title={\n                      \"text\": \"<b>Train/Test Month Distrubution</b> <BR />Train-Mars / Test--April-Oct<br> <br> \",\n                      \"x\":0.035,\n                      \"font_size\": 20,\n                      \n                  },\n                  plot_bgcolor='beige',\n                  paper_bgcolor='beige'\n\n)         \nfig.update_xaxes(showticklabels=True, showtickprefix='none')","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:32:04.761895Z","iopub.execute_input":"2023-05-22T09:32:04.762380Z","iopub.status.idle":"2023-05-22T09:32:11.602865Z","shell.execute_reply.started":"2023-05-22T09:32:04.762334Z","shell.execute_reply":"2023-05-22T09:32:11.600744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**note:** train data in 2017 has density of 4M, meanwhile, 2018 to 2019 has around 6M. ","metadata":{}},{"cell_type":"markdown","source":"? Is the density amount of data around oct 2018 to apr 2019 due to subsampling?","metadata":{}},{"cell_type":"code","source":"gc.collect()\nweek_days ={1: 'Mon', 2: 'Tue', 3: 'Wen', 4: 'Thu', 5: 'Fri', 6: 'Sat', 7: 'Sun'}\nfig = make_subplots(rows=1, cols=2)\n\ntrain_data['day_of_week'] = train_data['S_2'].apply(lambda x : x.isocalendar()[-1])\ntarget=pd.DataFrame(data={'Default': train_data.groupby(['day_of_week'])['target'].sum()})\ntarget['count'] = train_data['day_of_week'].value_counts()\ntarget.index = target.index.map(mapper=(lambda x: week_days[x]))\ntarget['Paid']= target['count'] - target['Default']\ntarget[\"perc_default\"] = (target['Default']*100)/target['count']\ntarget[\"perc_paid\"] = 100 - target[\"perc_default\"]\n\nfig.add_trace(go.Bar(x=target.index, y=target.Paid, name='Paid',\n                     text=target.perc_paid, texttemplate='%{text:.0f}%', \n                     textposition='inside', insidetextanchor=\"middle\",\n                     marker_color='#8cc7b5',\n                     hovertemplate=\"<b>%{x}</b><br>Paid accounts: %{text:.2f}%\"),\n              row=1, col=1)\n\nfig.add_trace(go.Bar(x=target.index, y=target.Default, name='Default',\n                     text=target.perc_default, texttemplate='%{text:.0f}%', \n                     textposition='inside', insidetextanchor=\"middle\",\n                     marker_color='#ecb6b6',\n                     hovertemplate=\"<b>%{x}</b><br>Default accounts: %{text:.2f}%\"),\n              row=1, col=1)\n\ntest_data['day_of_week'] = test_data['S_2'].apply(lambda x : x.isocalendar()[-1])\ndata = test_data.reset_index().groupby('day_of_week')['customer_ID'].count()\ndata.index = data.index.map(mapper=(lambda x: week_days[x]))\nfig.add_trace(go.Bar(x=data.index, y=data.values, name='Test', marker_color='#f6e0b5'),\n              row=1, col=2)\n\nfig.update_layout(template='ggplot2', \n                  title={\n                      \"text\": \"<b>Train/Test Weekdays Distribution</b><br><sub>Top1 = Saturday</sub>\",\n                      \"x\":0.05,\n                      \"font_size\": 18,\n                  },\n                  barmode='relative', yaxis_ticksuffix='%',\n                  margin={'pad':20},\n                  plot_bgcolor='#f5f5dc', # set background color\n                  showlegend=False, # hide legend\n                  )\n\n# add box around subplots\nfig.update_xaxes(showline=True, linewidth=1, linecolor='gray', mirror=True)\nfig.update_yaxes(showline=True, linewidth=1, linecolor='gray', mirror=True)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:32:42.263958Z","iopub.execute_input":"2023-05-22T09:32:42.265341Z","iopub.status.idle":"2023-05-22T09:34:50.857607Z","shell.execute_reply.started":"2023-05-22T09:32:42.265268Z","shell.execute_reply":"2023-05-22T09:34:50.856204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the plot, it appears that there is a higher density of records falling into the 'Paid' category on Saturdays compared to other days of the week. On the other hand, there is a lower density of records falling into the 'Default' category on Sundays compared to other days of the week.\n\nThis suggests that the **day of the week might be an important feature** in predicting the target variable. For example, it's possible that people are more likely to make their payments on Saturdays, or that there are fewer default events that happen on Sundays.\n\n*Update* = Customer statements probably drastically increase in Saturday, since end of the week is Saturday in USA.","metadata":{}},{"cell_type":"code","source":"gc.collect()\n# Then, group train and test data by 'S_2' and count unique customer IDs\ntrain_plot = train_data.reset_index().groupby('S_2')['customer_ID'].nunique()\ntest_plot = test_data.reset_index().groupby('S_2')['customer_ID'].nunique()\n\n# Create a subplot figure with two columns\nfig = make_subplots(rows=1, cols=2)\n\n# Add a scatter trace to the first column with the train data\nfig.add_trace(\n    go.Scatter(\n        x=train_plot.index, \n        y=train_plot.values, \n        mode='lines',\n        name='train',\n        line=dict(color=pastel_pink, width=3)\n    ),\n    row=1, col=1\n)\n\n# Add a scatter trace to the second column with the test data\nfig.add_trace(\n    go.Scatter(\n        x=test_plot.index, \n        y=test_plot.values, \n        mode='lines',\n        name='test',\n        line=dict(color=pastel_green, width=3)\n    ),\n    row=1, col=2\n)\n\n# Update the layout of the figure\nfig.update_layout(\n    hovermode=\"x unified\", \n    height=550,\n    margin={\"pad\":15},\n    title={\n        \"text\": \"<b>Train/Test Daily Customer Statements Count</b> <BR />Seasonal Trend on train data<br> <br> \",\n        \"x\":0.035,},\n    xaxis_title='Date',\n    yaxis_title='Customer Count'\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:35:02.970111Z","iopub.execute_input":"2023-05-22T09:35:02.970560Z","iopub.status.idle":"2023-05-22T09:35:13.607946Z","shell.execute_reply.started":"2023-05-22T09:35:02.970517Z","shell.execute_reply":"2023-05-22T09:35:13.606888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Investigation of Weekday's effect on numerical values**","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(20, 8))\n\nsns.boxplot(x=train_data.day_of_week, y=train_data.P_2.values, ax=ax, palette=['#F9B5AC', '#A3D9B1'])\n\nax.tick_params(left=False, bottom=False, labelsize=12)\nax.xaxis.label.set_fontsize(14)\nax.set_ylabel('D_42', fontsize=14)\n\n# Set background color\nax.set_facecolor('#F5F5DC')  # Beige background color\n\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])\n\n# Set figure background color\nfig.patch.set_facecolor('#F5F5DC')  # Beige background color\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:54:55.337463Z","iopub.execute_input":"2023-05-22T09:54:55.337941Z","iopub.status.idle":"2023-05-22T09:54:57.002539Z","shell.execute_reply.started":"2023-05-22T09:54:55.337897Z","shell.execute_reply":"2023-05-22T09:54:57.001153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(10, 6))\n\nsns.boxplot(x=train_data.day_of_week, y=train_data.S_11.values, ax=ax, palette=['#F9B5AC', '#A3D9B1'])\n\nax.tick_params(left=False, bottom=False, labelsize=12)\nax.xaxis.label.set_fontsize(14)\nax.set_ylabel('D_42', fontsize=14)\n\n# Set background color\nax.set_facecolor('#F5F5DC')  # Beige background color\n\nsns.despine(bottom=True, trim=True)\nplt.tight_layout(rect=[0, 0.2, 1, 0.99])\n\n# Set figure background color\nfig.patch.set_facecolor('#F5F5DC')  # Beige background color\n\n# Enable zooming\nplt.subplots_adjust(left=0.1, bottom=0.2)  # Adjust the margins for zooming\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T10:07:37.714302Z","iopub.execute_input":"2023-05-22T10:07:37.714814Z","iopub.status.idle":"2023-05-22T10:07:40.240387Z","shell.execute_reply.started":"2023-05-22T10:07:37.714771Z","shell.execute_reply":"2023-05-22T10:07:40.238906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***Density Plots for different Categories***\n\nSeaborKernel Density Estimate (KDE) Plot allows to dress the “shape” of our features, as a kind of continuous replacement for the discrete histogram. We will apply KDE for each category features, and try to see the density(amount), also another plot for time series.","metadata":{}},{"cell_type":"code","source":"def show_kdeplots(letter, figsize,plot_name):\n    # Filter columns based on conditions\n    cat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n    cols = [c for c in train_data.columns if c.startswith(letter) and c not in cat_cols and c not in ['S_2', 'customer_ID']]\n    df_tmp = train_data.sample(n=5000, random_state=42)[cols + ['target']].astype('float64')\n    plt_cols = 5\n    plt_rows = math.ceil(len(cols) / plt_cols)\n    \n    # Calculate height ratio for taller subplots\n    height_ratio = 1.5\n\n    # Set color palette\n    colors = ['#F9B5AC', '#A3D9B1']  # Pastel pink and pastel green\n    sns.set_palette(sns.color_palette(colors))\n\n    # Create the figure and axes\n    fig, axes = plt.subplots(plt_rows, plt_cols, figsize=(figsize[0], figsize[1]*height_ratio))\n\n    for i, ax in enumerate(axes.reshape(-1)):\n        if i < len(cols) - 1:\n            column_data = df_tmp[cols[i]]\n            if column_data.nunique() > 1:  # Check if column has more than one unique value\n                sns.kdeplot(x=cols[i], hue='target', hue_order=[1, 0], label=['Default', 'Paid'], data=df_tmp,\n                            fill=True, linewidth=2, legend=False, ax=ax, warn_singular=False)\n\n        ax.tick_params(left=False, bottom=False, labelsize=5)\n        ax.xaxis.label.set_fontsize(10)\n        ax.set_ylabel('')\n        \n        # Set background color\n        ax.set_facecolor('#F5F5DC')  # Beige background color\n\n    sns.despine(bottom=True, trim=True)\n    plt.tight_layout(rect=[0, 0.2, 1, 0.99])\n\n    # Set figure background color\n    fig.patch.set_facecolor('#F5F5DC')  # Beige background color\n    \n    # Add title\n    fig.suptitle(f'{plot_name} Density Plot', y=1.02)\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:35:57.113038Z","iopub.execute_input":"2023-05-22T09:35:57.114030Z","iopub.status.idle":"2023-05-22T09:35:57.133016Z","shell.execute_reply.started":"2023-05-22T09:35:57.113975Z","shell.execute_reply":"2023-05-22T09:35:57.131570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nshow_kdeplots('D',(20,16),\"Deliquency\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:36:00.978832Z","iopub.execute_input":"2023-05-22T09:36:00.979257Z","iopub.status.idle":"2023-05-22T09:36:22.549471Z","shell.execute_reply.started":"2023-05-22T09:36:00.979219Z","shell.execute_reply":"2023-05-22T09:36:22.548152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nshow_kdeplots('S',(10,8),\"Spend\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:37:36.064177Z","iopub.execute_input":"2023-05-22T09:37:36.064625Z","iopub.status.idle":"2023-05-22T09:37:44.146744Z","shell.execute_reply.started":"2023-05-22T09:37:36.064588Z","shell.execute_reply":"2023-05-22T09:37:44.141757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nshow_kdeplots('P',(20,4),\"Payment\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:38:17.464400Z","iopub.execute_input":"2023-05-22T09:38:17.465772Z","iopub.status.idle":"2023-05-22T09:38:19.370349Z","shell.execute_reply.started":"2023-05-22T09:38:17.465701Z","shell.execute_reply":"2023-05-22T09:38:19.369041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nshow_kdeplots('B',(20,16), \"Balance\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:38:35.151584Z","iopub.execute_input":"2023-05-22T09:38:35.152938Z","iopub.status.idle":"2023-05-22T09:38:45.600515Z","shell.execute_reply.started":"2023-05-22T09:38:35.152867Z","shell.execute_reply":"2023-05-22T09:38:45.599245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nshow_kdeplots('R',(16,16),\"Risk\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:42:54.592181Z","iopub.execute_input":"2023-05-22T09:42:54.592727Z","iopub.status.idle":"2023-05-22T09:43:02.375956Z","shell.execute_reply.started":"2023-05-22T09:42:54.592682Z","shell.execute_reply":"2023-05-22T09:43:02.374628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Customers Analysis**\n\nCheck for few customers how their values for each category change over time,\nalso pay attention to the customers exists only 1 time in dataset.","metadata":{}},{"cell_type":"markdown","source":"***Customer Statements Amount per Customer***","metadata":{}},{"cell_type":"code","source":"gc.collect()\n#Define pastel colors\ncolors = ['#FFC8C8', '#FFD8A8', '#FFE8C8', '#FFF8C8', '#F8FFC8', '#C8FFC8', '#C8FFF8']\n\n# Compute the statement counts per customer for train and test data\ntrain_counts = train_data['customer_ID'].value_counts().value_counts().sort_index(ascending=False).rename('Train statements per customer')\ntest_counts = test_data['customer_ID'].value_counts().value_counts().sort_index(ascending=False).rename('Test statements per customer')\n\n# Create pie chart for train data\ntrain_fig = go.Figure(data=[go.Pie(labels=train_counts.index.astype(str),\n                                    values=train_counts.values,\n                                    textposition='inside',\n                                    hole=0.5,\n                                    marker=dict(colors=px.colors.qualitative.Pastel1))])\ntrain_fig.update_layout(title='Train Data Statement Counts per Customer',\n                        margin=dict(l=0, r=0, t=30, b=0),\n                        paper_bgcolor='beige',\n                        plot_bgcolor='beige')\ntrain_fig.show()\n\n# Create pie chart for test data\ntest_fig = go.Figure(data=[go.Pie(labels=test_counts.index.astype(str),\n                                   values=test_counts.values,\n                                   textposition='inside',\n                                   hole=0.5,\n                                   marker=dict(colors=px.colors.qualitative.Pastel1))])\ntest_fig.update_layout(title='Test Data Statement Counts per Customer',\n                       margin=dict(l=0, r=0, t=30, b=0),\n                       paper_bgcolor='beige',\n                       plot_bgcolor='beige')\ntest_fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:45:24.754362Z","iopub.execute_input":"2023-05-22T09:45:24.754935Z","iopub.status.idle":"2023-05-22T09:45:28.504401Z","shell.execute_reply.started":"2023-05-22T09:45:24.754885Z","shell.execute_reply":"2023-05-22T09:45:28.503037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n\n# Filter customers with statement counts less than 13\ntrain_data_filtered = train_data.groupby('customer_ID').filter(lambda x: len(x) < 13)\ntest_data_filtered = test_data.groupby('customer_ID').filter(lambda x: len(x) < 13)\n\n# Compute the statement counts per customer for train and test data\ntrain_counts = train_data_filtered['customer_ID'].value_counts().value_counts().sort_index(ascending=False).rename('Train statements per customer')\ntest_counts = test_data_filtered['customer_ID'].value_counts().value_counts().sort_index(ascending=False).rename('Test statements per customer')\n\nfig = make_subplots(rows=1, cols=2)\n\n# Create bar chart for train data\ntrain_data_month = train_data_filtered.reset_index().groupby(['year', 'month'])['customer_ID'].count()\ntrain_data_month.index = train_data_month.index.to_series().apply(lambda x: f'{x[0]}-{x[1]}').values\nfig.add_trace(go.Bar(x=train_data_month.index, y=train_data_month.values, name='train', marker_color='#F5B7B1'), row=1, col=1)\n\n# Create bar chart for test data\ntest_data_month = test_data_filtered.reset_index().groupby(['year', 'month'])['customer_ID'].count()\ntest_data_month.index = test_data_month.index.to_series().apply(lambda x: f'{x[0]}-{x[1]}').values\nfig.add_trace(go.Bar(x=test_data_month.index, y=test_data_month.values, name='test', marker_color='#82E0AA'), row=1, col=2)\n\nfig.update_layout(template='ggplot2',\n                  title={\n                      \"text\": \"<b>Train/Test Month Distribution (Statement Count < 13)</b> <BR />Train-Mars / Test--April-Oct<br> <br> \",\n                      \"x\":0.035,\n                      \"font_size\": 20,\n                  },\n                  plot_bgcolor='beige',\n                  paper_bgcolor='beige',\n                  xaxis_tickprefix='',\n                  xaxis2_tickprefix=''\n)         \nfig.update_xaxes(showticklabels=True)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:45:37.421051Z","iopub.execute_input":"2023-05-22T09:45:37.421919Z","iopub.status.idle":"2023-05-22T09:49:47.235074Z","shell.execute_reply.started":"2023-05-22T09:45:37.421870Z","shell.execute_reply":"2023-05-22T09:49:47.233615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n\n# Filter customers with statement counts less than 13\ntrain_data_filtered = train_data.groupby('customer_ID').filter(lambda x: len(x) == 13)\ntest_data_filtered = test_data.groupby('customer_ID').filter(lambda x: len(x) == 13)\n\n# Compute the statement counts per customer for train and test data\ntrain_counts = train_data_filtered['customer_ID'].value_counts().value_counts().sort_index(ascending=False).rename('Train statements per customer')\ntest_counts = test_data_filtered['customer_ID'].value_counts().value_counts().sort_index(ascending=False).rename('Test statements per customer')\n\nfig = make_subplots(rows=1, cols=2)\n\n# Create bar chart for train data\ntrain_data_month = train_data_filtered.reset_index().groupby(['year', 'month'])['customer_ID'].count()\ntrain_data_month.index = train_data_month.index.to_series().apply(lambda x: f'{x[0]}-{x[1]}').values\nfig.add_trace(go.Bar(x=train_data_month.index, y=train_data_month.values, name='train', marker_color='#F5B7B1'), row=1, col=1)\n\n# Create bar chart for test data\ntest_data_month = test_data_filtered.reset_index().groupby(['year', 'month'])['customer_ID'].count()\ntest_data_month.index = test_data_month.index.to_series().apply(lambda x: f'{x[0]}-{x[1]}').values\nfig.add_trace(go.Bar(x=test_data_month.index, y=test_data_month.values, name='test', marker_color='#82E0AA'), row=1, col=2)\n\nfig.update_layout(template='ggplot2',\n                  title={\n                      \"text\": \"<b>Train/Test Month Distribution (Statement Count == 13)</b> <BR />Train-Mars / Test--April-Oct<br> <br> \",\n                      \"x\":0.035,\n                      \"font_size\": 20,\n                  },\n                  plot_bgcolor='beige',\n                  paper_bgcolor='beige',\n                  xaxis_tickprefix='',\n                  xaxis2_tickprefix=''\n)         \nfig.update_xaxes(showticklabels=True)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:50:10.972429Z","iopub.execute_input":"2023-05-22T09:50:10.972862Z","iopub.status.idle":"2023-05-22T09:54:16.353641Z","shell.execute_reply.started":"2023-05-22T09:50:10.972824Z","shell.execute_reply":"2023-05-22T09:54:16.352119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n\n# Filter customers with statement counts less than 13\ntrain_data_filtered = train_data.groupby('customer_ID').filter(lambda x: len(x) == 1)\ntest_data_filtered = test_data.groupby('customer_ID').filter(lambda x: len(x) == 1)\n\n# Compute the statement counts per customer for train and test data\ntrain_counts = train_data_filtered['customer_ID'].value_counts().value_counts().sort_index(ascending=False).rename('Train statements per customer')\ntest_counts = test_data_filtered['customer_ID'].value_counts().value_counts().sort_index(ascending=False).rename('Test statements per customer')\n\nfig = make_subplots(rows=1, cols=2)\n\n# Create bar chart for train data\ntrain_data_month = train_data_filtered.reset_index().groupby(['year', 'month'])['customer_ID'].count()\ntrain_data_month.index = train_data_month.index.to_series().apply(lambda x: f'{x[0]}-{x[1]}').values\nfig.add_trace(go.Bar(x=train_data_month.index, y=train_data_month.values, name='train', marker_color='#F5B7B1'), row=1, col=1)\n\n# Create bar chart for test data\ntest_data_month = test_data_filtered.reset_index().groupby(['year', 'month'])['customer_ID'].count()\ntest_data_month.index = test_data_month.index.to_series().apply(lambda x: f'{x[0]}-{x[1]}').values\nfig.add_trace(go.Bar(x=test_data_month.index, y=test_data_month.values, name='test', marker_color='#82E0AA'), row=1, col=2)\n\nfig.update_layout(template='ggplot2',\n                  title={\n                      \"text\": \"<b>Train/Test Month Distribution (Statement Count == 1)</b> <BR />Train-Mars / Test--April-Oct<br> <br> \",\n                      \"x\":0.035,\n                      \"font_size\": 20,\n                  },\n                  plot_bgcolor='beige',\n                  paper_bgcolor='beige',\n                  xaxis_tickprefix='',\n                  xaxis2_tickprefix=''\n)         \nfig.update_xaxes(showticklabels=True)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T09:56:02.037174Z","iopub.execute_input":"2023-05-22T09:56:02.037704Z","iopub.status.idle":"2023-05-22T09:59:39.217258Z","shell.execute_reply.started":"2023-05-22T09:56:02.037662Z","shell.execute_reply":"2023-05-22T09:59:39.215776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_temp = train_data.S_2.groupby(train_data.customer_ID).max()\ntest_temp = test_data.S_2.groupby(test_data.customer_ID).max()\n\n# Define subplots\nfig = make_subplots(rows=1,cols=2, subplot_titles=(\"Train Set\",\"Test Set\"))\nfig.add_trace(go.Histogram(x=train_temp, xbins=dict(start='2017-03-01', end='2018-03-31',size='D'), \n                            marker_color='#ffd700', opacity=0.75,),\n             row=1, col=1)\nfig.add_trace(go.Histogram(x=test_temp, xbins=dict(start='2018-04-01', end='2019-11-01',size='D'), \n                            marker_color=pastel_purple, opacity=0.75,),\n             row=1, col=2)\n# Set layout\nfig.update_layout(\n    template='ggplot2',\n    title={\n        \"text\": \"<b>Train/Test Month Distribution</b> <BR />Train-March / Test-April-Oct<br> <br> \",\n        \"x\":0.035,\n        \"font_size\": 20,\n    },\n    xaxis_title='Last statement date per customer',\n    yaxis_title='Count',\n    plot_bgcolor='beige',\n    paper_bgcolor='beige',\n    showlegend=False,\n    bargap=0.05,\n    bargroupgap=0.2,\n)\n\n# Show the plot\nfig.show()\n\n# Delete temporary variables\ndel train_temp, test_temp","metadata":{"execution":{"iopub.status.busy":"2023-05-22T10:00:18.282891Z","iopub.execute_input":"2023-05-22T10:00:18.283387Z","iopub.status.idle":"2023-05-22T10:01:12.534749Z","shell.execute_reply.started":"2023-05-22T10:00:18.283340Z","shell.execute_reply":"2023-05-22T10:01:12.532769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Categorical Values**","metadata":{}},{"cell_type":"code","source":"gc.collect()\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\ndef count_plot(col, title):\n    # Compute target percentage by group\n    target = pd.DataFrame({'Default': train_data.groupby([col])['target'].mean() * 100})\n    target['Paid'] = 100 - target['Default']\n\n    # Create subplots\n    fig = make_subplots(rows=1, cols=2)\n\n    # Add Paid bar chart\n    fig.add_trace(go.Bar(x=target.index, y=target.Paid, name='Paid',\n                         text=target.Paid, texttemplate='%{text:.0f}%', \n                         textposition='inside', insidetextanchor=\"middle\",\n                         marker=dict(color='#8cbf86', line=dict(color='#8cbf86', width=0)),\n                         hovertemplate=\"<b>%{x}</b> | Paid accounts: %{y:.2f}%\"),\n                  row=1, col=1)\n\n    # Add Default bar chart\n    fig.add_trace(go.Bar(x=target.index, y=target.Default, name='Default',\n                         text=target.Default, texttemplate='%{text:.0f}%', \n                         textposition='inside', insidetextanchor=\"middle\",\n                         marker=dict(color='#f9a8b8', line=dict(color='#f9a8b8', width=0)),\n                         hovertemplate=\"<b>%{x}</b> | Default accounts: %{y:.2f}%\"),\n                  row=1, col=1)\n\n    # Add Test bar chart\n    data = test_data[col].value_counts()\n    fig.add_trace(go.Bar(x=data.index, y=data.values, name='Test', \n                         marker=dict(color='#f3d3bd', line=dict(color='#f3d3bd', width=0))),\n                  row=1, col=2)\n\n    # Update layout\n    fig.update_layout(template='plotly_white', title={\n                        \"text\": f\"<b>{title}</b>\",\n                        \"x\": 0.035,\n                        \"font_size\": 18},\n                      plot_bgcolor='#f9ecec', paper_bgcolor='#f9ecec',\n                      barmode='relative', yaxis_ticksuffix='%',\n                      height=550,\n                      legend=dict(orientation=\"h\", traceorder=\"reversed\", yanchor=\"bottom\", y=1.05, xanchor=\"left\", x=0.9))\n\n    # Show figure\n    fig.show()\n\n# Call the function for each categorical column\nfor col in cat_cols:\n    count_plot(col, f'Train/Test {col} Counts')","metadata":{"execution":{"iopub.status.busy":"2023-05-22T10:35:40.972049Z","iopub.execute_input":"2023-05-22T10:35:40.973813Z","iopub.status.idle":"2023-05-22T10:35:43.582184Z","shell.execute_reply.started":"2023-05-22T10:35:40.973751Z","shell.execute_reply":"2023-05-22T10:35:43.580736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Correlation Analysis** ","metadata":{}},{"cell_type":"code","source":"corr = train_data.corrwith(train_data['target'], axis=0)\ncorr = corr[corr.notna()].sort_values(ascending=False)\n\npos = corr[(corr > 0) & (corr < 1)]\nneg = corr[corr < 0]\n\nfig = go.Figure()\nfig.add_trace(go.Bar(\n    x=pos.index,\n    y=pos.values,\n    orientation='v',\n    name='Positive Correlation',\n    marker=dict(color='#ffb6c1', line=dict(color='#c9a0dc', width=0)),\n    text=[\"%.2f\" % (round(v, 2) * 100) + '%' for v in pos.values],\n    textposition='outside',\n    textfont_color='#4E1C1E'\n))\n\nfig.add_trace(go.Bar(\n    x=neg.index,\n    y=neg.values,\n    orientation='v',\n    name='Negative Correlation',\n    marker=dict(color='#98fb98', line=dict(color='#ffb6c1', width=0)),\n    text=[\"%.2f\" % (round(v, 2) * 100) + '%' for v in neg.values],\n    textposition='outside',\n    textfont_color='#4E1C1E'\n))\n\nfig.update_layout(\n    template='plotly_white',\n    title={\n        \"text\": \"<b>Pearson Correlation with The Payment Default Feature</b> <BR />Extreme elements are the top correlated features<br> <br> \",\n        \"x\": 0.035,\n        \"font_size\": 18,\n    },\n    plot_bgcolor='#f9ecec',\n    paper_bgcolor='#f9ecec',\n    legend=dict(\n        y=1.18,\n        x=0.88\n    )\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T10:40:51.673853Z","iopub.execute_input":"2023-05-22T10:40:51.674402Z","iopub.status.idle":"2023-05-22T10:41:14.069417Z","shell.execute_reply.started":"2023-05-22T10:40:51.674356Z","shell.execute_reply":"2023-05-22T10:41:14.067908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate correlation matrix\ncorr_matrix = train_data.sample(n=10000, random_state=42).corr()\n\n# Create heatmap\nfig = go.Figure(data=go.Heatmap(\n    z=corr_matrix.values,\n    x=corr_matrix.columns,\n    y=corr_matrix.columns,\n    colorscale='Viridis',\n    zmin=-1,\n    zmax=1))\n\nfig.update_layout(\n    title='Correlation Heatmap',\n    xaxis_title='Features',\n    yaxis_title='Features',\n    plot_bgcolor='beige')\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T10:59:42.398700Z","iopub.execute_input":"2023-05-22T10:59:42.399194Z","iopub.status.idle":"2023-05-22T10:59:43.541234Z","shell.execute_reply.started":"2023-05-22T10:59:42.399148Z","shell.execute_reply":"2023-05-22T10:59:43.539788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Noise Detection**","metadata":{}},{"cell_type":"code","source":"def plot_noise(col, threshold, bins=400):\n    fig, axes = plt.subplots(1, 3, figsize=(20, 5))\n    colors = [\"beige\", \"lightgreen\", \"lightcoral\"]\n    hue_order = [0, 1]\n\n    for i, ax in enumerate(axes):\n        if i == 0:\n            sns.histplot(data=train_data, x=col, hue=\"target\", bins=bins, ax=ax)\n        elif i == 1:\n            sns.histplot(data=train_data[train_data[col] < threshold], x=col, hue=\"target\", bins=bins, ax=ax)\n        elif i == 2:\n            sns.histplot(data=train_data[train_data[col] > threshold], x=col, hue=\"target\", bins=bins, ax=ax)\n\n        ax.set_title(f\"Noise in {col}\")\n        ax.set_xlabel(\"Value\")\n        ax.set_ylabel(\"Frequency\")\n        ax.set_facecolor(\"beige\")\n        ax.legend([\"Class 0\", \"Class 1\"], loc=\"upper right\")\n        ax.get_legend().remove()\n\n        # Set colors\n        ax.patches[0].set_facecolor(\"darkgreen\")\n        ax.patches[1].set_facecolor(\"lightcoral\")\n        ax.spines[\"bottom\"].set_color(\"gray\")\n        ax.spines[\"top\"].set_color(\"gray\")\n        ax.spines[\"right\"].set_color(\"gray\")\n        ax.spines[\"left\"].set_color(\"gray\")\n        ax.tick_params(axis=\"x\", colors=\"gray\")\n        ax.tick_params(axis=\"y\", colors=\"gray\")\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:00:37.291398Z","iopub.execute_input":"2023-05-22T11:00:37.291892Z","iopub.status.idle":"2023-05-22T11:00:37.307750Z","shell.execute_reply.started":"2023-05-22T11:00:37.291845Z","shell.execute_reply":"2023-05-22T11:00:37.306372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_noise(col = \"B_33\", threshold = 0.2 )","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:00:42.970025Z","iopub.execute_input":"2023-05-22T11:00:42.970518Z","iopub.status.idle":"2023-05-22T11:01:03.750094Z","shell.execute_reply.started":"2023-05-22T11:00:42.970472Z","shell.execute_reply":"2023-05-22T11:01:03.748646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_noise(\"D_48\",threshold = 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:01:09.960142Z","iopub.execute_input":"2023-05-22T11:01:09.960630Z","iopub.status.idle":"2023-05-22T11:01:28.808303Z","shell.execute_reply.started":"2023-05-22T11:01:09.960586Z","shell.execute_reply":"2023-05-22T11:01:28.807120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_noise(\"B_9\",threshold = 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:02:16.550922Z","iopub.execute_input":"2023-05-22T11:02:16.551497Z","iopub.status.idle":"2023-05-22T11:02:37.012163Z","shell.execute_reply.started":"2023-05-22T11:02:16.551450Z","shell.execute_reply":"2023-05-22T11:02:37.009310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_noise(\"D_44\",threshold = 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:04:28.470910Z","iopub.execute_input":"2023-05-22T11:04:28.471523Z","iopub.status.idle":"2023-05-22T11:04:48.159359Z","shell.execute_reply.started":"2023-05-22T11:04:28.471472Z","shell.execute_reply":"2023-05-22T11:04:48.157973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_noise(\"D_75\",threshold = 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:05:57.802376Z","iopub.execute_input":"2023-05-22T11:05:57.802871Z","iopub.status.idle":"2023-05-22T11:06:17.585709Z","shell.execute_reply.started":"2023-05-22T11:05:57.802830Z","shell.execute_reply":"2023-05-22T11:06:17.584150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_noise(\"D_55\",threshold = 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:08:15.698696Z","iopub.execute_input":"2023-05-22T11:08:15.699202Z","iopub.status.idle":"2023-05-22T11:08:36.260734Z","shell.execute_reply.started":"2023-05-22T11:08:15.699154Z","shell.execute_reply":"2023-05-22T11:08:36.259349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_noise(\"D_58\",threshold = 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:10:14.963844Z","iopub.execute_input":"2023-05-22T11:10:14.964420Z","iopub.status.idle":"2023-05-22T11:10:35.148577Z","shell.execute_reply.started":"2023-05-22T11:10:14.964374Z","shell.execute_reply":"2023-05-22T11:10:35.147214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_noise(\"S_11\",threshold = 0.2)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:27:15.911256Z","iopub.execute_input":"2023-05-22T11:27:15.911840Z","iopub.status.idle":"2023-05-22T11:27:36.313941Z","shell.execute_reply.started":"2023-05-22T11:27:15.911795Z","shell.execute_reply":"2023-05-22T11:27:36.312422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Null Features Analysis**","metadata":{}},{"cell_type":"code","source":"gc.collect()\ntotal = train_data.isna().sum().sort_values(ascending = False)\npercent = ((train_data.isna().sum()/train_data.isna().count()*100).sort_values(ascending = False))\nmissing_data = pd.concat([total,percent],axis=1,keys=['Total Missing','Percentage Missing'])","metadata":{"execution":{"iopub.status.busy":"2023-05-21T12:26:09.657969Z","iopub.execute_input":"2023-05-21T12:26:09.658498Z","iopub.status.idle":"2023-05-21T12:26:29.664147Z","shell.execute_reply.started":"2023-05-21T12:26:09.658458Z","shell.execute_reply":"2023-05-21T12:26:29.662519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n\n# Generate pastel colors\nnum_features = len(missing_data)\ncolors = ['hsl(' + str(h) + ',80%' + ',80%)' for h in np.linspace(0, 360, num_features)]\n\n# Create the figure\nfig = go.Figure()\n\n# Add the bar trace\nfig.add_trace(go.Bar(\n    x=missing_data['Percentage Missing'].sort_values(ascending=True),\n    y=missing_data['Percentage Missing'].sort_values(ascending=True).index,\n    marker=dict(color=colors),\n    orientation='h',\n    name='',\n    hovertemplate='%{y} Percentage Missing of the feature: %{x}',\n    showlegend=False\n))\n\n# Update layout\nfig.update_layout(\n    title=\"Percentage of Missing Data of Each Feature\",\n    xaxis_title=\"Percentage\",\n    margin=dict(l=150),\n    height=3000,\n    width=700,\n    hovermode='closest',\n    paper_bgcolor='beige',\n    plot_bgcolor='beige'\n)\n\n# Show the plot\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T10:29:24.054276Z","iopub.execute_input":"2023-05-21T10:29:24.056007Z","iopub.status.idle":"2023-05-21T10:29:24.984861Z","shell.execute_reply.started":"2023-05-21T10:29:24.055934Z","shell.execute_reply":"2023-05-21T10:29:24.983758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\n\n# Select features with missing data more than 50%\nmissing_data_50 = missing_data[missing_data['Percentage Missing'] > 50]\nfeatures = missing_data_50.index.tolist()\n\n# Create a new dataframe with selected features and the target variable\ncorrelation_df = train_data[features + ['target']]\n\n# Calculate the correlation matrix\ncorrelation_matrix = correlation_df.corr()\n\n# Set the style and color palette\nsns.set_theme(style=\"white\")\ncolors = ['#F9B5AC', '#A3D9B1']  # Pastel pink and pastel green\nsns.set_palette(sns.color_palette(colors))\n\n# Create the heatmap\nplt.figure(figsize=(12, 10), facecolor='beige')\nax = sns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True, square=True,\n                 annot_kws={\"size\": 8}, linewidths=0.5, linecolor='lightgray')\nax.set_facecolor('beige')\n\nplt.title('Correlation Heatmap: Features with Missing Data > 50%')\nplt.tight_layout()\n\n# Enable interactive mode\nplt.ion()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T10:40:33.779408Z","iopub.execute_input":"2023-05-21T10:40:33.779926Z","iopub.status.idle":"2023-05-21T10:40:46.613392Z","shell.execute_reply.started":"2023-05-21T10:40:33.779885Z","shell.execute_reply":"2023-05-21T10:40:46.611622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA Notes**\n* Customer statements increase in the saturday, drops on monday. We can say that day of week might be an important feature for the most of the data and it is worth to do an EDA focusing on day of week.\n","metadata":{}}]}