{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom sklearn.metrics import roc_auc_score\nimport math\nimport scipy\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\nplt.rc(\"font\", size=14)\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:43:26.938716Z","iopub.execute_input":"2022-08-09T08:43:26.939575Z","iopub.status.idle":"2022-08-09T08:43:28.177625Z","shell.execute_reply.started":"2022-08-09T08:43:26.939518Z","shell.execute_reply":"2022-08-09T08:43:28.176736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this kernel I will provide you how to get the most important features using gini coefficient, weight of evidence(WOE) and information value(IV).\n\nI create 1418 variables and select the top 25 from them","metadata":{}},{"cell_type":"code","source":"# Cutoffs for features selection\n\nNull_Share_Cutoff = 0.8 # cutoff for features with many Null\nNUnique_Cutoff = 1 # cutoff for constant features\nGini_Cutoff = 0.4 # individual gini cutoff\nCorr_Cutoff = 0.6 # correlation cutoff\nIV_Cutoff = 0.85 # information value cutoff","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:08:50.195642Z","iopub.execute_input":"2022-08-09T09:08:50.196241Z","iopub.status.idle":"2022-08-09T09:08:50.201660Z","shell.execute_reply.started":"2022-08-09T09:08:50.196192Z","shell.execute_reply":"2022-08-09T09:08:50.200851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndata_row = pd.read_parquet(\"../input/amex-parquet/train_data.parquet\")\ntarget_sample = pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:43:28.187237Z","iopub.execute_input":"2022-08-09T08:43:28.188545Z","iopub.status.idle":"2022-08-09T08:44:07.807905Z","shell.execute_reply.started":"2022-08-09T08:43:28.188499Z","shell.execute_reply":"2022-08-09T08:44:07.806673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_row.drop(['target'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:44:07.810460Z","iopub.execute_input":"2022-08-09T08:44:07.810834Z","iopub.status.idle":"2022-08-09T08:44:10.684675Z","shell.execute_reply.started":"2022-08-09T08:44:07.810782Z","shell.execute_reply":"2022-08-09T08:44:10.683613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\ndef func(x):\n    x1 = int(''.join([x2 for x2 in x if x2 !='-']))\n    return x1\n\ndata_row.S_2 = data_row.S_2.apply(lambda x: func(x))\ndata_row[[\"S_2\"]] = MinMaxScaler().fit_transform(data_row[[\"S_2\"]])\n\n#Divide the variable D_63 into two groups\ndata_row['D_631'] = data_row['D_63'].apply(lambda x: x[0])\ndata_row['D_632'] = data_row['D_63'].apply(lambda x: x[1])\n\nprint(data_row.D_63.unique())\nprint(data_row.D_64.unique())\nprint(data_row.D_631.unique())\nprint(data_row.D_632.unique())","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:44:10.686159Z","iopub.execute_input":"2022-08-09T08:44:10.686546Z","iopub.status.idle":"2022-08-09T08:44:25.453333Z","shell.execute_reply.started":"2022-08-09T08:44:10.686511Z","shell.execute_reply":"2022-08-09T08:44:25.452114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68', 'D_631', 'D_632']\nnumerical =  [x for x in data_row.columns[1:] if x not in categorical]\n\nDICT_num = dict.fromkeys(numerical, ['mean', 'min', 'max', 'std', 'last', \"first\", 'sum']) \nDICT_cat = dict.fromkeys(categorical, ['nunique', 'first', 'last'])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:44:25.455063Z","iopub.execute_input":"2022-08-09T08:44:25.455906Z","iopub.status.idle":"2022-08-09T08:44:25.463320Z","shell.execute_reply.started":"2022-08-09T08:44:25.455864Z","shell.execute_reply":"2022-08-09T08:44:25.462138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# aggregate data using dicts above\n\ndata_num = data_row.groupby(['customer_ID'], as_index=False).agg(DICT_num)\ndata_cat = data_row.groupby(['customer_ID'], as_index=False).agg(DICT_cat)\n\ndata_psm = data_row.groupby(['customer_ID'], as_index=False)[categorical].agg(lambda x: str(scipy.stats.mode(x)[0][0]))\n\ndata_num.columns = ['customer_ID'] + ['_'.join(col) for col in data_num.columns if col[0] != 'customer_ID']\ndata_cat.columns = ['customer_ID'] + ['_'.join(col) for col in data_cat.columns if col[0] != 'customer_ID']\ndata_psm.columns = ['customer_ID'] + [ col + '_psm' for col in data_psm.columns if col != 'customer_ID']","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:44:25.465280Z","iopub.execute_input":"2022-08-09T08:44:25.465759Z","iopub.status.idle":"2022-08-09T08:57:06.831553Z","shell.execute_reply.started":"2022-08-09T08:44:25.465709Z","shell.execute_reply":"2022-08-09T08:57:06.828911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del [data_row]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:57:06.835818Z","iopub.execute_input":"2022-08-09T08:57:06.836626Z","iopub.status.idle":"2022-08-09T08:57:07.104183Z","shell.execute_reply.started":"2022-08-09T08:57:06.836569Z","shell.execute_reply":"2022-08-09T08:57:07.102906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# columns for get dummies\nfor_gd = [col for col in data_psm.columns if col != 'customer_ID'] + [col for col in data_cat.columns \n                                                                if (col != 'customer_ID' and col[-7:] != 'nunique')]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:57:07.106185Z","iopub.execute_input":"2022-08-09T08:57:07.107032Z","iopub.status.idle":"2022-08-09T08:57:07.116641Z","shell.execute_reply.started":"2022-08-09T08:57:07.106981Z","shell.execute_reply":"2022-08-09T08:57:07.115761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Combine into one table\n\ndata = data_num.merge(data_cat, on='customer_ID')\ndata = data.merge(data_psm, on='customer_ID')\n\ndata = pd.get_dummies(data, columns=for_gd)\n\ndata = data.merge(target_sample, on='customer_ID')\ndata.fillna(0, inplace=True)\nvariables = data.columns[1:-1] #columns without customer_ID and target","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:01:04.910361Z","iopub.execute_input":"2022-08-09T09:01:04.911020Z","iopub.status.idle":"2022-08-09T09:01:32.531362Z","shell.execute_reply.started":"2022-08-09T09:01:04.910981Z","shell.execute_reply":"2022-08-09T09:01:32.529965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of varianles:', len(variables))","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:01:32.534297Z","iopub.execute_input":"2022-08-09T09:01:32.534811Z","iopub.status.idle":"2022-08-09T09:01:32.541575Z","shell.execute_reply.started":"2022-08-09T09:01:32.534763Z","shell.execute_reply":"2022-08-09T09:01:32.540436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we collect information about variables for their selection","metadata":{}},{"cell_type":"code","source":"%%time\n\ndef stats_for_var(x):\n    if x.dtype=='object':\n        return pd.Series([x.isnull().sum(),x.isnull().sum()/len(x),x.nunique()])\n    else:\n        return pd.Series([x.isnull().sum(),x.isnull().sum()/len(x),x.nunique(),x.min(),x.max(),x.mean(),x.std()])\n\ndef gini_for_var(x):\n    gini = abs(roc_auc_score(data.target.values,data[x])*2-1)   \n    return gini\n\n\nstats = data[variables].apply(stats_for_var).T\nstats.columns=['null','null_share','nunique','min','max','mean','std']\nstats['gini'] = stats.index.map(gini_for_var)\n\nstats.sort_values(by='gini',ascending=False).head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:01:32.543072Z","iopub.execute_input":"2022-08-09T09:01:32.543455Z","iopub.status.idle":"2022-08-09T09:05:49.303972Z","shell.execute_reply.started":"2022-08-09T09:01:32.543417Z","shell.execute_reply":"2022-08-09T09:05:49.302843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_drop=stats[['null_share','nunique','gini']].copy()\n\n# Using Cutoffs for drop bad variables\nto_drop['null_share']=np.where(to_drop['null_share'] >= Null_Share_Cutoff, 1, 0)\nto_drop['nunique']=np.where(to_drop['nunique'] <= NUnique_Cutoff, 1, 0)\nto_drop['gini']=np.where(to_drop['gini'] <= Gini_Cutoff, 1, 0)\n\nto_drop = to_drop[to_drop.eq(1).any(axis=1)==1]\n\nprint(len(to_drop),'variables dropped:')\nvariables1 = [i for i in variables if i not in to_drop.index.values]\nprint(len(variables1),'variables left')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:05:49.307251Z","iopub.execute_input":"2022-08-09T09:05:49.307842Z","iopub.status.idle":"2022-08-09T09:05:49.365015Z","shell.execute_reply.started":"2022-08-09T09:05:49.307806Z","shell.execute_reply":"2022-08-09T09:05:49.363836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n\ndf_corr = data[variables1].corr()\ndf_corr = df_corr.abs()\n\ncorrelations = df_corr.where(np.triu(\n        np.ones(df_corr.shape), k=1\n    ).astype(np.bool)).stack().sort_values(ascending = False)\n\nsns.histplot(correlations, bins=100)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:05:49.366551Z","iopub.execute_input":"2022-08-09T09:05:49.367306Z","iopub.status.idle":"2022-08-09T09:06:59.922743Z","shell.execute_reply.started":"2022-08-09T09:05:49.367263Z","shell.execute_reply":"2022-08-09T09:06:59.921537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop variables with big correlations\n# if the variables are correlated we drop the one with less gini\n\nvar_to_drop_list = []\nfor i in correlations[correlations > Corr_Cutoff].index:\n    v1 = i[0]\n    v2 = i[1]\n    \n    if (v1 in var_to_drop_list or v2 in var_to_drop_list):\n        continue    \n\n    var_to_drop = stats.loc[[v1,v2]].gini.idxmin()  #variable with less gini\n    var_to_drop_list.append(var_to_drop)\n    \nprint(len(var_to_drop_list), 'variables dropped')\nvariables2 = [i for i in variables1 if i not in var_to_drop_list]\nprint(len(variables2),'variables left')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:06:59.924347Z","iopub.execute_input":"2022-08-09T09:06:59.924770Z","iopub.status.idle":"2022-08-09T09:07:00.071533Z","shell.execute_reply.started":"2022-08-09T09:06:59.924734Z","shell.execute_reply":"2022-08-09T09:07:00.070405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def iv_for_var(data):\n    \n    data['share'] = data['all_cnt'] / data['all_cnt'].sum()\n    data['distribution_of_good'] = data['target_cnt'] / data['target_cnt'].sum()\n    data['distribution_of_bad'] = (data['all_cnt'] - data['target_cnt']) / (data['all_cnt'].sum() - data['target_cnt'].sum())\n    data['woe'] = np.log(data['distribution_of_good'] / data['distribution_of_bad'])\n\n    data['iv'] = data['woe'] * (data['distribution_of_good'] - data['distribution_of_bad'])\n    \n    return data['iv'].sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:07:00.072774Z","iopub.execute_input":"2022-08-09T09:07:00.073091Z","iopub.status.idle":"2022-08-09T09:07:00.080253Z","shell.execute_reply.started":"2022-08-09T09:07:00.073062Z","shell.execute_reply":"2022-08-09T09:07:00.079195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Weight of Evidence (WoE) and Information Value (IV) can be used to understand the predictive power of an independent variable. Who helps to understand if a particular class of an independent variable has a higher distribution of good or bad.\n\nTo begin with, let's bin numerical variables","metadata":{}},{"cell_type":"code","source":"vars_to_bin = data[variables2].nunique()\nvars_to_bin = vars_to_bin[vars_to_bin > 10].index # variables that we ишт\nvars_info_iv = pd.DataFrame(columns=['variable','iv_train'])\n\n\nfor var in variables2:\n    if var in vars_to_bin:\n        var_bin = pd.qcut(data[var], q=15, duplicates='drop') # Bin it into 15 parts\n    else:\n        var_bin = data[var]\n        \n    df_var_bin = pd.DataFrame({'bucket': var_bin, 'target': data['target']})\n    \n    var_bin_grp = df_var_bin.groupby(['bucket']).agg({'bucket':'count','target':'sum'}).rename(\n    columns={'bucket':'all_cnt', 'target':'target_cnt'}\n        ).reset_index(level=0)\n    \n    var_iv = iv_for_var(var_bin_grp)\n    vars_info_iv = vars_info_iv.append(\n        pd.DataFrame([[var,var_iv]], columns=['variable','iv_train']), ignore_index = True)\n    \nvars_info_iv","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:07:00.081812Z","iopub.execute_input":"2022-08-09T09:07:00.082206Z","iopub.status.idle":"2022-08-09T09:07:03.712943Z","shell.execute_reply.started":"2022-08-09T09:07:00.082175Z","shell.execute_reply":"2022-08-09T09:07:03.711796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Using cutoff for IV\ncondition = (vars_info_iv['iv_train'] < IV_Cutoff)\nprint('low IV variables dropped:',len(vars_info_iv[condition]))\n\nto_drop = vars_info_iv[condition]['variable'].values.tolist()\n\nvariables_final = [var for var in variables2 if var not in to_drop]\nprint('variables left:',len(variables_final))","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:08:54.888426Z","iopub.execute_input":"2022-08-09T09:08:54.889780Z","iopub.status.idle":"2022-08-09T09:08:54.899240Z","shell.execute_reply.started":"2022-08-09T09:08:54.889729Z","shell.execute_reply":"2022-08-09T09:08:54.898253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we select top 25 features and we can say that they will be useful. Selecting features based on IV and Gini may result in excluding a relevant feature from the model.","metadata":{}},{"cell_type":"code","source":"print(variables_final)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T09:09:14.225885Z","iopub.execute_input":"2022-08-09T09:09:14.226304Z","iopub.status.idle":"2022-08-09T09:09:14.231833Z","shell.execute_reply.started":"2022-08-09T09:09:14.226272Z","shell.execute_reply":"2022-08-09T09:09:14.230960Z"},"trusted":true},"execution_count":null,"outputs":[]}]}