{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nsns.set_style(\"darkgrid\", {\"grid.color\": \".6\", \"grid.linestyle\": \":\"})\nfrom tqdm.auto import tqdm\nimport plotly.express as px\nimport dask.dataframe as dd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-16T09:53:53.845349Z","iopub.execute_input":"2022-06-16T09:53:53.845764Z","iopub.status.idle":"2022-06-16T09:53:53.855363Z","shell.execute_reply.started":"2022-06-16T09:53:53.845729Z","shell.execute_reply":"2022-06-16T09:53:53.854582Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CONTENT TABLE\n----------------\n* [1) Introduction](#01)\n* [2) Variables Groups](#02)\n* [3) Null Variables](#03)\n* [4) Types of Variable](#04)\n* [5) Correlation of Variables](#05)\n* [6) Analysing groups correlation](#06)\n    * [6.1) D Variables](#06.1)\n    * [6.2) S Variables](#06.2)\n    * [6.3) P Variables](#06.3)\n    * [6.4) R Variables](#06.4)\n    * [6.5) B Variables](#06.5)\n* [7) Summary](#07)\n  ","metadata":{}},{"cell_type":"markdown","source":"<a id=\"01\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Introduction </p>\n<div style=\"font-family: courier; font-size:20px\">\n<li>This notebook is a brief analysis of the train dataset from amex. It was used the original dataset, without any transformations in variables. ( it is highly recommended one of the many post related to reducing the data in the discussions)\n    \n<li>To read the data, I used the dask.dataframe library and the main plots were done with plotly. Other libraries recommended to cope with the huge amount of data are the cudf and dask_cudf. <br>\n\n<li>The main observations in this notebook consists of null values\\variables in training dataset and the correlation between variables, target as well.\n</div>","metadata":{}},{"cell_type":"code","source":"df = dd.read_csv('../input/amex-default-prediction/train_data.csv')\ny = dd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:11:12.507422Z","iopub.execute_input":"2022-06-16T10:11:12.508273Z","iopub.status.idle":"2022-06-16T10:11:12.549680Z","shell.execute_reply.started":"2022-06-16T10:11:12.508230Z","shell.execute_reply":"2022-06-16T10:11:12.548854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Total number of registers {df.shape[0].compute()}, with {df.shape[1]} columns')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:53:57.519947Z","iopub.execute_input":"2022-06-16T09:53:57.520499Z","iopub.status.idle":"2022-06-16T09:57:18.604098Z","shell.execute_reply.started":"2022-06-16T09:53:57.520463Z","shell.execute_reply":"2022-06-16T09:57:18.603237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"02\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Variables Groups </p>\n<div style=\"font-family: courier; font-size:20px\">\n    Separaing the variables according to the competion data description  <br><br>\n        <b>Variables</b> \n        <li>D_* = Delinquency variables</li>\n        <li>S_* = Spend variables</li>\n        <li>P_* = Payment variables</li>\n        <li>B_* = Balance variables</li>\n        <li>R_* = Risk variables</li>\n</div>","metadata":{}},{"cell_type":"code","source":"#Function to separate each variable into its category\ndef variables(cols, verbose = True):\n    vars_groups = {'S':[],\"D\":[],\n            \"B\":[],\"R\":[],\"P\":[], 'idx':[]}\n\n\n    for c in cols:\n    \n        if 'customer_ID' == c:\n            vars_groups['idx'].append(c)\n        elif \"S\" in c:\n            vars_groups['S'].append(c)\n        elif \"D\" in c:\n            vars_groups['D'].append(c)\n        elif \"B\" in c:\n            vars_groups['B'].append(c)\n        elif \"R\" in c:\n            vars_groups['R'].append(c)\n        elif \"P\" in c:\n            vars_groups['P'].append(c)\n\n\n\n    if verbose:\n        print(' Groups:', vars_groups.keys(),'\\n','Number of Groups:', len(vars_groups))\n    return vars_groups","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:59:03.962921Z","iopub.execute_input":"2022-06-16T09:59:03.963278Z","iopub.status.idle":"2022-06-16T09:59:03.977111Z","shell.execute_reply.started":"2022-06-16T09:59:03.963248Z","shell.execute_reply":"2022-06-16T09:59:03.976320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vars_groups = variables(df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:59:06.176845Z","iopub.execute_input":"2022-06-16T09:59:06.177491Z","iopub.status.idle":"2022-06-16T09:59:06.182252Z","shell.execute_reply.started":"2022-06-16T09:59:06.177451Z","shell.execute_reply":"2022-06-16T09:59:06.181045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of variables for each categories ')\ntable = {'Group':[], 'Total':[]}\nfor g, i in vars_groups.items():\n    table['Group'].append(g)\n    table['Total'].append(len(i))\n    print(g, len(i), ' variables')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:59:14.659207Z","iopub.execute_input":"2022-06-16T09:59:14.659818Z","iopub.status.idle":"2022-06-16T09:59:14.672274Z","shell.execute_reply.started":"2022-06-16T09:59:14.659777Z","shell.execute_reply":"2022-06-16T09:59:14.671049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"table_df = pd.DataFrame(table).sort_values(by = 'Total', ascending = False)\n\nfig = px.bar(table_df, \n            x=\"Group\", \n            y=\"Total\", \n            color = 'Group',\n            text = 'Total',\n            width=800,\n            height=600)\n\nfig.update_layout(legend=dict(\n    orientation=\"h\",\n    yanchor=\"bottom\",\n    y=1.02,\n    xanchor=\"right\",\n    x=0.9,\n    title= 'Types of Variables'\n))","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:59:49.004572Z","iopub.execute_input":"2022-06-16T09:59:49.005610Z","iopub.status.idle":"2022-06-16T09:59:49.974706Z","shell.execute_reply.started":"2022-06-16T09:59:49.005542Z","shell.execute_reply":"2022-06-16T09:59:49.973957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"03\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Null Variables </p>\n<div style=\"font-family: courier; font-size:20px\">\n</div>","metadata":{}},{"cell_type":"code","source":"nulls = df.isnull().mean().compute().sort_values(ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:00:53.288182Z","iopub.execute_input":"2022-06-16T10:00:53.288534Z","iopub.status.idle":"2022-06-16T10:03:56.555902Z","shell.execute_reply.started":"2022-06-16T10:00:53.288500Z","shell.execute_reply":"2022-06-16T10:03:56.555067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thr = 0.6\nfig = px.bar(nulls,\n        color = nulls.values > thr,\n        title  = f'Percentage of Nulls | Red-> higher than {thr}% of nulls | With {len(nulls[nulls > thr])}/{len(nulls)} variables above threshold',\n        color_discrete_map={1: 'red', 0:'blue'})\n\nfig.update_layout(legend=dict(\n    orientation=\"h\",\n    yanchor=\"bottom\",\n    y=1.02,\n    xanchor=\"right\",\n    x=0.9,\n    title= 'Above Threshold'\n))\nfig.show()\nnulls_var = list(nulls[nulls >thr].index)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:05:39.703217Z","iopub.execute_input":"2022-06-16T10:05:39.703720Z","iopub.status.idle":"2022-06-16T10:05:39.849892Z","shell.execute_reply.started":"2022-06-16T10:05:39.703686Z","shell.execute_reply":"2022-06-16T10:05:39.849098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"04\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Types of Variable </p>\n<div style=\"font-family: courier; font-size:20px\">\n\n</div>","metadata":{}},{"cell_type":"code","source":"vars_type ={}\nfor i in df.columns:\n    vars_type[i] =df[i].dtype\n\ndf_table =  pd.DataFrame(vars_type.values(), columns = ['types']).value_counts().reset_index()\ndf_table['types'] = df_table['types'].apply(str)\ndf_table.rename(columns = {0:'Count'}, inplace = True)\n\nfig =px.bar(df_table, \n            x = 'types', \n            y = 'Count',\n            text = 'Count',\n            color ='types',\n            width=800, height=400)\n\n\nfig.update_layout(legend=dict(\n    orientation=\"h\",\n    yanchor=\"bottom\",\n    y=1.02,\n    xanchor=\"right\",\n    x=0.9,\n    title= 'Types of Variables'\n))\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:05:52.517227Z","iopub.execute_input":"2022-06-16T10:05:52.517825Z","iopub.status.idle":"2022-06-16T10:05:52.613202Z","shell.execute_reply.started":"2022-06-16T10:05:52.517787Z","shell.execute_reply":"2022-06-16T10:05:52.612396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"05\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Correlation of Variables </p>\n<div style=\"font-family: courier; font-size:20px\">\n    <li> Group by customer ID and get the last information\n    <li> Merge target column to the grouped data\n</div>","metadata":{}},{"cell_type":"code","source":"df_grouped = df.groupby('customer_ID').last().compute()\ny_grouped = y.groupby('customer_ID').last().compute()\ndf_grouped = dd.merge(df_grouped, y_grouped, on = 'customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:06:20.004115Z","iopub.execute_input":"2022-06-16T10:06:20.004462Z","iopub.status.idle":"2022-06-16T10:10:07.452158Z","shell.execute_reply.started":"2022-06-16T10:06:20.004433Z","shell.execute_reply":"2022-06-16T10:10:07.446737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = df_grouped.corr()['target']\ncorr = corr.reset_index()\nplt.figure(figsize = (20,30))\nsns.barplot(y = 'index', x = 'target',data = corr.sort_values('target',ascending = False)[1:], palette = 'magma')\nplt.title('Variables Correlation to the Target')","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:14:51.219828Z","iopub.execute_input":"2022-06-16T10:14:51.220553Z","iopub.status.idle":"2022-06-16T10:15:28.099359Z","shell.execute_reply.started":"2022-06-16T10:14:51.220509Z","shell.execute_reply":"2022-06-16T10:15:28.098584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"font-family: courier; font-size:20px\">\n    <li> Top positive and negative correlated variables to the target\n</div>","metadata":{}},{"cell_type":"code","source":"top10_pos = corr.sort_values('target',ascending = False)[1:][:10]\ntop10_neg = corr.sort_values('target',ascending = False)[1:][-10:]\nbest_cor_vars = [*top10_pos['index'], *top10_neg['index']]\nprint('Top 10 positive Correlation with target\\n',top10_pos,'\\n')\nprint('Top 10 negative Correlation with target\\n', top10_neg)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:17:51.754036Z","iopub.execute_input":"2022-06-16T10:17:51.754808Z","iopub.status.idle":"2022-06-16T10:17:51.767999Z","shell.execute_reply.started":"2022-06-16T10:17:51.754766Z","shell.execute_reply":"2022-06-16T10:17:51.767007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"06\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Analysing groups correlation </p>\n<div style=\"font-family: courier; font-size:20px\">\nIn this section it is going to be performed a analysis for each one of the groups \n</div>","metadata":{}},{"cell_type":"markdown","source":"<a id=\"06.1\"></a>\n# <p style=\"background-color:#002663;height: 30px;text-align: center;vertical-align: middle;line-height: 30px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> D Group </p>","metadata":{}},{"cell_type":"code","source":"df_v = df_grouped[vars_groups['D']]","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:18:20.331469Z","iopub.execute_input":"2022-06-16T10:18:20.332084Z","iopub.status.idle":"2022-06-16T10:18:20.913273Z","shell.execute_reply.started":"2022-06-16T10:18:20.332046Z","shell.execute_reply":"2022-06-16T10:18:20.912415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thr_c = 0.6\ndef correlation_filter(df_v, thr_c = thr_c):\n    c = df_v.corr().abs().unstack()\n    so = c.sort_values(ascending = False, kind=\"quicksort\")\n    vars_removed = []\n    m = pd.DataFrame(so)\n    m.rename(columns ={0:'corr_value'}, inplace = True)\n    m1 = m[m['corr_value']<1].reset_index()\n    var_corr = m1[m1['corr_value'] >thr_c].drop_duplicates(subset = ['level_0','level_1']).groupby('level_0')['level_1'].apply(list).reset_index(name='list')\n    var_corr['len'] = var_corr['list'].apply(len)\n\n    to_stay = []\n    to_remove = []\n    for i, var in enumerate(var_corr.level_0):\n        for var_2 in var_corr.iloc[i].list:\n            if (var_2 not in best_cor_vars) & ( var_2 not in to_stay):\n                to_remove.append(var_2)\n            \n        to_stay.append(var)\n\n    to_remove = list(set(to_remove))\n    other_Vars = [i for i in df_v.columns if i not in to_remove]\n    return var_corr, to_remove,other_Vars","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:18:52.064954Z","iopub.execute_input":"2022-06-16T10:18:52.065522Z","iopub.status.idle":"2022-06-16T10:18:52.076011Z","shell.execute_reply.started":"2022-06-16T10:18:52.065487Z","shell.execute_reply":"2022-06-16T10:18:52.074489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_d, vars_filter_d, vars_filtered_d = correlation_filter(df_v)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:19:00.246337Z","iopub.execute_input":"2022-06-16T10:19:00.246910Z","iopub.status.idle":"2022-06-16T10:19:08.949795Z","shell.execute_reply.started":"2022-06-16T10:19:00.246873Z","shell.execute_reply":"2022-06-16T10:19:08.948978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_d","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:19:23.938235Z","iopub.execute_input":"2022-06-16T10:19:23.939195Z","iopub.status.idle":"2022-06-16T10:19:23.966433Z","shell.execute_reply.started":"2022-06-16T10:19:23.939146Z","shell.execute_reply":"2022-06-16T10:19:23.965451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig =px.bar(var_corr_d.sort_values(by = 'len', ascending = False), \n            x = 'level_0', \n            y = 'len',\n            text = 'len',\n            color ='level_0',\n            title = f'Number of correlateded variables in D above threshold of {thr_c}',\n            width=1800, height=600)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:20:03.816240Z","iopub.execute_input":"2022-06-16T10:20:03.816953Z","iopub.status.idle":"2022-06-16T10:20:04.054586Z","shell.execute_reply.started":"2022-06-16T10:20:03.816914Z","shell.execute_reply":"2022-06-16T10:20:04.053724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"06.2\"></a>\n# <p style=\"background-color:#002663;height: 30px;text-align: center;vertical-align: middle;line-height: 30px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> S Group </p>","metadata":{}},{"cell_type":"code","source":"df_v = df_grouped[vars_groups['S']]","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:20:41.890498Z","iopub.execute_input":"2022-06-16T10:20:41.891507Z","iopub.status.idle":"2022-06-16T10:20:41.940899Z","shell.execute_reply.started":"2022-06-16T10:20:41.891461Z","shell.execute_reply":"2022-06-16T10:20:41.940006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_s, vars_filter_s, vars_filtered_s= correlation_filter(df_v)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:20:48.807041Z","iopub.execute_input":"2022-06-16T10:20:48.807942Z","iopub.status.idle":"2022-06-16T10:20:49.370247Z","shell.execute_reply.started":"2022-06-16T10:20:48.807902Z","shell.execute_reply":"2022-06-16T10:20:49.369423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_s","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:20:55.649764Z","iopub.execute_input":"2022-06-16T10:20:55.650127Z","iopub.status.idle":"2022-06-16T10:20:55.660852Z","shell.execute_reply.started":"2022-06-16T10:20:55.650098Z","shell.execute_reply":"2022-06-16T10:20:55.660111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig =px.bar(var_corr_s.sort_values(by = 'len', ascending = False), \n            x = 'level_0', \n            y = 'len',\n            text = 'len',\n            color ='level_0',\n            title = f'Number of correlateded variables in S above threshold of {thr_c}',\n            width=1800, height=600)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:21:05.059280Z","iopub.execute_input":"2022-06-16T10:21:05.059641Z","iopub.status.idle":"2022-06-16T10:21:05.142238Z","shell.execute_reply.started":"2022-06-16T10:21:05.059611Z","shell.execute_reply":"2022-06-16T10:21:05.141499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"06.3\"></a>\n# <p style=\"background-color:#002663;height: 30px;text-align: center;vertical-align: middle;line-height: 30px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> P Group </p>","metadata":{}},{"cell_type":"code","source":"df_v = df_grouped[vars_groups['P']]","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:21:18.118318Z","iopub.execute_input":"2022-06-16T10:21:18.118782Z","iopub.status.idle":"2022-06-16T10:21:18.128190Z","shell.execute_reply.started":"2022-06-16T10:21:18.118747Z","shell.execute_reply":"2022-06-16T10:21:18.127244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_p, vars_filter_p,vars_filtered_p = correlation_filter(df_v)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:21:25.066147Z","iopub.execute_input":"2022-06-16T10:21:25.066845Z","iopub.status.idle":"2022-06-16T10:21:25.097815Z","shell.execute_reply.started":"2022-06-16T10:21:25.066805Z","shell.execute_reply":"2022-06-16T10:21:25.097034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_p\n","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:21:35.370602Z","iopub.execute_input":"2022-06-16T10:21:35.371278Z","iopub.status.idle":"2022-06-16T10:21:35.381247Z","shell.execute_reply.started":"2022-06-16T10:21:35.371240Z","shell.execute_reply":"2022-06-16T10:21:35.380254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"06.4\"></a>\n# <p style=\"background-color:#002663;height: 30px;text-align: center;vertical-align: middle;line-height: 30px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> R Group </p>","metadata":{}},{"cell_type":"code","source":"df_v = df_grouped[vars_groups['R']]","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:21:47.213995Z","iopub.execute_input":"2022-06-16T10:21:47.214729Z","iopub.status.idle":"2022-06-16T10:21:47.254525Z","shell.execute_reply.started":"2022-06-16T10:21:47.214693Z","shell.execute_reply":"2022-06-16T10:21:47.253614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_r, vars_filter_r, vars_filtered_r= correlation_filter(df_v)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:21:54.355950Z","iopub.execute_input":"2022-06-16T10:21:54.356302Z","iopub.status.idle":"2022-06-16T10:21:55.215058Z","shell.execute_reply.started":"2022-06-16T10:21:54.356272Z","shell.execute_reply":"2022-06-16T10:21:55.214267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_r","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:22:07.628690Z","iopub.execute_input":"2022-06-16T10:22:07.629674Z","iopub.status.idle":"2022-06-16T10:22:07.644591Z","shell.execute_reply.started":"2022-06-16T10:22:07.629637Z","shell.execute_reply":"2022-06-16T10:22:07.643802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig =px.bar(var_corr_r.sort_values(by = 'len', ascending = False), \n            x = 'level_0', \n            y = 'len',\n            text = 'len',\n            color ='level_0',\n            title = f'Number of correlateded variables in R above threshold of {thr_c}',\n            width=1800, height=600)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:22:12.234248Z","iopub.execute_input":"2022-06-16T10:22:12.234625Z","iopub.status.idle":"2022-06-16T10:22:12.329781Z","shell.execute_reply.started":"2022-06-16T10:22:12.234587Z","shell.execute_reply":"2022-06-16T10:22:12.328921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"06.5\"></a>\n# <p style=\"background-color:#002663;height: 30px;text-align: center;vertical-align: middle;line-height: 30px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> B Group </p>","metadata":{}},{"cell_type":"code","source":"df_v = df_grouped[vars_groups['B']]\n","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:22:25.349087Z","iopub.execute_input":"2022-06-16T10:22:25.349911Z","iopub.status.idle":"2022-06-16T10:22:25.399266Z","shell.execute_reply.started":"2022-06-16T10:22:25.349871Z","shell.execute_reply":"2022-06-16T10:22:25.398376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_b, vars_filter_b,vars_filtered_b = correlation_filter(df_v)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:22:36.234947Z","iopub.execute_input":"2022-06-16T10:22:36.235693Z","iopub.status.idle":"2022-06-16T10:22:37.914510Z","shell.execute_reply.started":"2022-06-16T10:22:36.235656Z","shell.execute_reply":"2022-06-16T10:22:37.913702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var_corr_b","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:22:51.802396Z","iopub.execute_input":"2022-06-16T10:22:51.802761Z","iopub.status.idle":"2022-06-16T10:22:51.820013Z","shell.execute_reply.started":"2022-06-16T10:22:51.802729Z","shell.execute_reply":"2022-06-16T10:22:51.819231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig =px.bar(var_corr_b.sort_values(by = 'len', ascending = False), \n            x = 'level_0', \n            y = 'len',\n            text = 'len',\n            color ='level_0',\n            title = f'Number of correlateded variables in B above threshold of {thr_c}',\n            width=1800, height=600)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:22:47.723102Z","iopub.execute_input":"2022-06-16T10:22:47.723471Z","iopub.status.idle":"2022-06-16T10:22:47.873370Z","shell.execute_reply.started":"2022-06-16T10:22:47.723440Z","shell.execute_reply":"2022-06-16T10:22:47.872519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"07\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Summary </p>\n","metadata":{}},{"cell_type":"code","source":"print('Best Correlation with target:\\n', best_cor_vars)\nprint(f'Above threshold of {thr} nulls:\\n',nulls_var)\n\nprint('\\nPossible Variables for each group\\n')\nprint('D)\\n\\tVariables to filter:\\n', vars_filter_d,'\\n\\tVariables to use:\\n',vars_filtered_d)\nprint('S)\\n\\tVariables to filter:\\n', vars_filter_s,'\\n\\tVariables to use:\\n',vars_filtered_s)\nprint('P)\\n\\tVariables to filter:\\n', vars_filter_p,'\\n\\tVariables to use:\\n',vars_filtered_p)\nprint('R)\\n\\tVariables to filter:\\n', vars_filter_r,'\\n\\tVariables to use:\\n',vars_filtered_r)\nprint('B)\\n\\tVariables to filter:\\n', vars_filter_b,'\\n\\tVariables to use:\\n',vars_filtered_b)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:23:42.907094Z","iopub.execute_input":"2022-06-16T10:23:42.907454Z","iopub.status.idle":"2022-06-16T10:23:42.915026Z","shell.execute_reply.started":"2022-06-16T10:23:42.907423Z","shell.execute_reply":"2022-06-16T10:23:42.914268Z"},"trusted":true},"execution_count":null,"outputs":[]}]}