{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This notebook explore different types of feature selection techniques \nBased on article share by the following link\nhttps://medium.com/analytics-vidhya/feature-selection-techniques-2614b3b7efcd\n\nFeature selection can be done in 3 categories:\n1. [Filter Method](#section-one)\n    - [Pearson Correlation](#subsection-one)\n    - [Variance Inflation Factor (VIF)](#subsection-two)\n1. [Wrapper Method](#section-two)\n    - [Step Forward Selection](#subsection-three)\n    - [Backward elimination](#subsection-four)\n    - [Recursive Feature elimination](#subsection-five)\n1. [Embedded Method](#section-three)\n1. [Summary](#section-four)","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport catboost as cat\nfrom catboost import CatBoostClassifier\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score,roc_curve,log_loss\nfrom sklearn.metrics import mean_squared_error\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-25T14:07:21.788510Z","iopub.execute_input":"2023-08-25T14:07:21.789395Z","iopub.status.idle":"2023-08-25T14:07:23.924573Z","shell.execute_reply.started":"2023-08-25T14:07:21.789268Z","shell.execute_reply":"2023-08-25T14:07:23.922869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainData = pd.read_csv('/kaggle/input/playground-series-s3e21/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:23.927878Z","iopub.execute_input":"2023-08-25T14:07:23.929869Z","iopub.status.idle":"2023-08-25T14:07:23.985102Z","shell.execute_reply.started":"2023-08-25T14:07:23.929813Z","shell.execute_reply":"2023-08-25T14:07:23.983945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainData","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:23.986860Z","iopub.execute_input":"2023-08-25T14:07:23.987560Z","iopub.status.idle":"2023-08-25T14:07:24.052530Z","shell.execute_reply.started":"2023-08-25T14:07:23.987522Z","shell.execute_reply":"2023-08-25T14:07:24.051121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Process trainData to fill up null, drop unused columns that are not present in test dataset","metadata":{}},{"cell_type":"code","source":"def generateXY(trainData):\n    x = trainData.drop(columns=['id','target'])\n    x = x.fillna(method='ffill')\n    x = x.fillna(0)\n    y = trainData.target\n    return x,y","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:24.056283Z","iopub.execute_input":"2023-08-25T14:07:24.056837Z","iopub.status.idle":"2023-08-25T14:07:24.063947Z","shell.execute_reply.started":"2023-08-25T14:07:24.056788Z","shell.execute_reply":"2023-08-25T14:07:24.062529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Calculate RMSE based on selected features","metadata":{}},{"cell_type":"code","source":"def calculateRMSE(x,y,features,model):\n    for i,val in enumerate(x.columns):\n        if val not in features:\n            x = x.drop(columns=val)\n    X_train, X_val, y_train, y_val = train_test_split(x,y,test_size=0.2, random_state=42)\n    clf = RandomForestRegressor(\n       n_estimators=1000,\n       max_depth=7,\n       n_jobs=-1,\n       random_state=42)\n    clf.fit(X_train,y_train)\n    prediction = clf.predict(X_val)\n    loss = mean_squared_error(y_val,prediction,squared=False)\n    print(f\"{model} has RMSE = {loss}\")\n    return clf","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:24.066056Z","iopub.execute_input":"2023-08-25T14:07:24.066888Z","iopub.status.idle":"2023-08-25T14:07:24.080970Z","shell.execute_reply.started":"2023-08-25T14:07:24.066840Z","shell.execute_reply":"2023-08-25T14:07:24.079116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# 1. Filter method\nFilter and take only the subset of the relevant features, most commonly done using Pearson correlation and VIF\n\n<a id=\"subsection-one\"></a>\n# A] Pearson Correlation\n\nThe Pearson correlation measures the strength of the linear relationship between two variables. It has a value between -1 to 1, with a value of -1 meaning a total negative linear correlation, 0 being no correlation, and + 1 meaning a total positive correlation.","metadata":{}},{"cell_type":"code","source":"trainCorr = trainData.corr()\nplt.figure(figsize=(30,15))\nsns.heatmap(trainCorr, annot=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:24.083438Z","iopub.execute_input":"2023-08-25T14:07:24.084314Z","iopub.status.idle":"2023-08-25T14:07:32.064261Z","shell.execute_reply.started":"2023-08-25T14:07:24.084263Z","shell.execute_reply":"2023-08-25T14:07:32.063328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Setting threshold to remove features with low correlation with target label (emission)","metadata":{}},{"cell_type":"code","source":"threshold = 0.05\ncorr=abs(trainCorr['target'])\nresult = corr[corr>threshold]\nresult.sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:32.065571Z","iopub.execute_input":"2023-08-25T14:07:32.066199Z","iopub.status.idle":"2023-08-25T14:07:32.078964Z","shell.execute_reply.started":"2023-08-25T14:07:32.066163Z","shell.execute_reply":"2023-08-25T14:07:32.078040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Discover features with high correlation, and keep and drop the other. Same process is followed until last variable.","metadata":{}},{"cell_type":"code","source":"columns = trainData.columns.tolist()\nhighCorrFeature = []\nfor i in range(5,len(columns)):\n    if columns[i] in highCorrFeature:\n        continue\n    for j in range(i,len(columns)):\n        if columns[j] in highCorrFeature:\n            continue\n        if i != j:\n            c = trainData[[columns[i],columns[j]]].corr()\n            val = c[columns[j]][0]\n            if val > 0.5:\n                print(f\"{columns[i]} and {columns[j]} has high correlation {val}\")\n                highCorrFeature.append(columns[j])\n                \nprint(f\"\\nSize of original data {trainData.shape[1]}, \\\nnumber of high correlation features {len(highCorrFeature)}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:32.080105Z","iopub.execute_input":"2023-08-25T14:07:32.081098Z","iopub.status.idle":"2023-08-25T14:07:32.511742Z","shell.execute_reply.started":"2023-08-25T14:07:32.081062Z","shell.execute_reply":"2023-08-25T14:07:32.510514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = trainData.drop(columns=highCorrFeature)\nfeaturePC = x.columns.tolist()\nprint(\"Selected features: \",featurePC)\nx,y = generateXY(trainData)\ncalculateRMSE(x,y,featurePC,'Pearson Correlation')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:32.513567Z","iopub.execute_input":"2023-08-25T14:07:32.514426Z","iopub.status.idle":"2023-08-25T14:07:39.279744Z","shell.execute_reply.started":"2023-08-25T14:07:32.514375Z","shell.execute_reply":"2023-08-25T14:07:39.278470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"subsection-two\"></a>\n# B] Variance Inflation Factor (VIF)\n\nVariance inflation factor (VIF) is a measure of the amount of multicollinearity in a set of multiple regression variables. \n\nTo detect collinearity among variables, simply create a correlation matrix and find variables with large absolute values.","metadata":{}},{"cell_type":"code","source":"from statsmodels.stats.outliers_influence import variance_inflation_factor\ndef cal_vif(x):\n    thresh = 2\n    output = pd.DataFrame()\n    k = x.shape[1]\n    vif = [variance_inflation_factor(x.values,i) for i in range(x.shape[1])]\n    for i in range(1,k):\n        a = np.argmax(vif)\n        print('Max vif is for variable no:',a)\n        if (vif[a]<=thresh):\n            break\n        if (i==1):\n            output = x.drop(x.columns[a],axis=1)\n            vif = [variance_inflation_factor(output.values,j) for j in range(output.shape[1])]\n        elif (i>1):\n            output = output.drop(output.columns[a],axis=1)\n            vif = [variance_inflation_factor(output.values,j) for j in range(output.shape[1])]\n    return output\n\nx,y = generateXY(trainData)\nx = cal_vif(x)\nfeatureVIF = x.columns.tolist()\nprint(\"Selected_features: \",featureVIF)\ncalculateRMSE(x,y,featureVIF,'Variance Inflation Factor')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:39.281597Z","iopub.execute_input":"2023-08-25T14:07:39.282978Z","iopub.status.idle":"2023-08-25T14:07:50.082293Z","shell.execute_reply.started":"2023-08-25T14:07:39.282913Z","shell.execute_reply":"2023-08-25T14:07:50.079940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-two\"></a>\n# 2. Wrapper Method\n\n<a id=\"subsection-three\"></a>\n# A] Step Forward Selection\n\nStep forward selection starts with the evaluation of each individual feature, and selects that which results in the best performing selected algorithm model. \n\nWe start with having no feature in the model. In each iteration, we keep adding the feature which best improves our model till an addition of a new variable does not improve the performance of the model.","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score as acc\nfrom mlxtend.feature_selection import SequentialFeatureSelector as sfs\nimport warnings\nwarnings.filterwarnings('ignore')\nx,y = generateXY(trainData)\nX_train, X_val, y_train, y_val = train_test_split(x,y,test_size=0.2, random_state=42)\nclf = CatBoostRegressor(iterations=10,metric_period=500,verbose=False)\nsfs1 = sfs(clf,k_features=5,forward=True,floating=False, verbose=1, cv=3)\nsfs1 = sfs1.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:07:50.087998Z","iopub.execute_input":"2023-08-25T14:07:50.088391Z","iopub.status.idle":"2023-08-25T14:08:15.143576Z","shell.execute_reply.started":"2023-08-25T14:07:50.088357Z","shell.execute_reply":"2023-08-25T14:08:15.142328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_cols = list(sfs1.k_feature_idx_)\nfeatureSFS = x.columns[feat_cols].tolist()\nprint(\"Selected features:\", featureSFS)\ncalculateRMSE(x,y,featureSFS,'Step Forward Selection')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:15.144957Z","iopub.execute_input":"2023-08-25T14:08:15.145317Z","iopub.status.idle":"2023-08-25T14:08:18.920181Z","shell.execute_reply.started":"2023-08-25T14:08:15.145284Z","shell.execute_reply":"2023-08-25T14:08:18.918944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"subsection-four\"></a>\n# B] Backward Elimination\n\nIn backward elimination, we start with all the features and removes the least significant feature at each iteration which improves the performance of the model. We repeat this until no improvement is observed on removal of features.","metadata":{}},{"cell_type":"code","source":"x,y = generateXY(trainData)\nimport statsmodels.api as sm\ncols = list(x.columns)\npmax = 1\nwhile (len(cols)>0):\n    p = []\n    x_1 = x[cols]\n    x_1 = sm.add_constant(x_1)\n    model = sm.OLS(y,x_1).fit()\n    p = pd.Series(model.pvalues.values[1:],index=cols)\n    pmax = max(p)\n    features_with_p_max = p.idxmax()\n    if (pmax>0.05):\n        cols.remove(features_with_p_max)\n    else:\n        break\nfeatureBE=cols\nprint(\"Selected features: \",featureBE)\ncalculateRMSE(x,y,featureBE,'Backward Elimination')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:18.921991Z","iopub.execute_input":"2023-08-25T14:08:18.922506Z","iopub.status.idle":"2023-08-25T14:08:24.437892Z","shell.execute_reply.started":"2023-08-25T14:08:18.922455Z","shell.execute_reply":"2023-08-25T14:08:24.436534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"subsection-five\"></a>\n# C] Recursive Feature elimination\n\nRecursive feature elimination (RFE) is a feature selection method that fits a model and removes the weakest feature (or features) until the specified number of features is reached.","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_selection import RFE\nx,y = generateXY(trainData)\nclf = CatBoostRegressor(iterations=100,metric_period=500,verbose=False)\nrfe = RFE(clf)\nX_rfe = rfe.fit_transform(x,y)\nclf.fit(X_rfe,y)\ncols = list(x.columns)\ntemp = pd.Series(rfe.support_,index = cols)\nfeatureRFE = temp[temp==True].index.tolist()\nprint(\"Selected features: \",featureRFE)\ncalculateRMSE(x,y,featureRFE,'Recursive Feature Elimination')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:24.439664Z","iopub.execute_input":"2023-08-25T14:08:24.440182Z","iopub.status.idle":"2023-08-25T14:08:36.876706Z","shell.execute_reply.started":"2023-08-25T14:08:24.440135Z","shell.execute_reply":"2023-08-25T14:08:36.875501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three\"></a>\n# 3. Embedded Method\n\nEmbedded methods are iterative process to extract those features which contribute the most to the training for a particular iteration. Regularization methods, such as Lasso regularization, is the most commonly used embedded methods which penalize a feature given a coefficient threshold. If the feature is irrelevant, lasso penalizes its coefficient and make it 0. Hence the features with coefficient = 0 are removed and the rest are taken.","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.linear_model import LassoCV\nx,y = generateXY(trainData)\nreg = LassoCV(cv=5,n_alphas=200,tol=1e-4,fit_intercept=False)\nreg.fit(x,y)\nprint(\"Best score: \", reg.score(x,y))\ncoef = pd.Series(reg.coef_, index=x.columns)\nprint(\"Lasso picked \" + str(sum(coef != 0))+ \" variables and eliminated the other \" + str(sum(coef ==0)) + \" variables\")\nimp_coef = coef.sort_values()\nfeatureEM = imp_coef[imp_coef>0].index.tolist()\nprint(\"Selected features: \",featureEM)\ncalculateRMSE(x,y,featureEM,'Embedded Method')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:36.878415Z","iopub.execute_input":"2023-08-25T14:08:36.879172Z","iopub.status.idle":"2023-08-25T14:08:42.233745Z","shell.execute_reply.started":"2023-08-25T14:08:36.879121Z","shell.execute_reply":"2023-08-25T14:08:42.232828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-four\"></a>\n# Summary\n\nFinal model is contructed based on the list of selected feature based on 6 different techniques:\n* Pearson Correlation\n* Variance Inflation Factor\n* Step Forward Selection\n* Backward elimination\n* Recursive Feature elimination\n* Embedded Method\n\n","metadata":{}},{"cell_type":"code","source":"allfeature = featurePC + featureVIF + featureSFS + featureBE + featureRFE + featureEM\nfeature, count = np.unique(allfeature, return_counts=True)\nfeatureCount = pd.DataFrame(count,index=feature, columns=['Count'])\nfeatureCount = featureCount.sort_values(by=['Count'])\nfeatureCount.plot(kind = \"barh\")\nplt.rcParams[\"figure.figsize\"] = (10,30)\nplt.title(\"Feature importance based on all selection techniques\")","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:42.235131Z","iopub.execute_input":"2023-08-25T14:08:42.235522Z","iopub.status.idle":"2023-08-25T14:08:42.901857Z","shell.execute_reply.started":"2023-08-25T14:08:42.235487Z","shell.execute_reply":"2023-08-25T14:08:42.900616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_features = []\nfor i, val in enumerate(feature):\n    if count[i] > 4:\n        selected_features.append(val)\nprint(\"Final selected features: \", selected_features)\ntrainData = pd.read_csv('/kaggle/input/playground-series-s3e21/sample_submission.csv')\nx,y = generateXY(trainData)\nclf = calculateRMSE(x,y,selected_features,\"Final model\")","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:42.903960Z","iopub.execute_input":"2023-08-25T14:08:42.904473Z","iopub.status.idle":"2023-08-25T14:08:46.528157Z","shell.execute_reply.started":"2023-08-25T14:08:42.904425Z","shell.execute_reply":"2023-08-25T14:08:46.526684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Replace dropped features with zeros","metadata":{}},{"cell_type":"code","source":"trainData = pd.read_csv('/kaggle/input/playground-series-s3e21/sample_submission.csv')\nfor i in trainData.columns:\n    if i not in selected_features and i != \"target\" and i != \"id\":\n        trainData[i] = 0\n        \nx,y = generateXY(trainData)\nclf = calculateRMSE(x,y,selected_features,\"Final model\")","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:46.529875Z","iopub.execute_input":"2023-08-25T14:08:46.530749Z","iopub.status.idle":"2023-08-25T14:08:50.144199Z","shell.execute_reply.started":"2023-08-25T14:08:46.530699Z","shell.execute_reply":"2023-08-25T14:08:50.142961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final Submission","metadata":{}},{"cell_type":"code","source":"trainData.to_csv('sample_submission.csv',index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:50.145569Z","iopub.execute_input":"2023-08-25T14:08:50.145932Z","iopub.status.idle":"2023-08-25T14:08:50.202785Z","shell.execute_reply.started":"2023-08-25T14:08:50.145898Z","shell.execute_reply":"2023-08-25T14:08:50.201537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainData","metadata":{"execution":{"iopub.status.busy":"2023-08-25T14:08:50.204367Z","iopub.execute_input":"2023-08-25T14:08:50.204736Z","iopub.status.idle":"2023-08-25T14:08:50.237864Z","shell.execute_reply.started":"2023-08-25T14:08:50.204703Z","shell.execute_reply":"2023-08-25T14:08:50.236815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![Thanks](data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAT4AAACfCAMAAABX0UX9AAABIFBMVEX////+/v4AAAAdHR3DiUAaGhr///0ZGRkXFxcREREQEBANDQ0UFBT7+/u5ubn///wnJyd4eHihoaHx8fGqqqrChzvq6urFxcXj4+OPj4/Y2NghISHQ0NCzs7Oenp5DQ0NcXFxsbGw5OTm/gi8tLS2JiYlQUFBzc3NCQkJVVVWVlZVkZGSBgYE0NDRtbW3U1NTy6+DNpXbo0LTIl1rMoGjcwZrYtofEjkrs4M/69u6+eyTbw6TFjk7m1sHbvJb59ujTsX3PuIq6gijGjj3Hll7QpHXClVDDhEHeu57o3dLm18XYto++eB/HuaPXvIzKmmXgxq8gEQC4iE6Ddm82KROpiWR0VixiTS5OOiJyZ1q3k2aieD2YajO9fTUCEBnTrIcQUcucAAAcSElEQVR4nO1dCXviSJKVUkgCJIE4xX0KMJfNjSkwru6yy13lnnZt7+zszM7W+v//i41MnYCwAaW7qc8dXx02RyrzZRwvIlNKhjldWJb88dHCjyF4jCz1YbIEu3eAHsu8AXoYuPeBHksdPNY03HeAnqEmNFtkTZf3DtCzfBTVBt+H2TJmdPwLvhPlTUyMpT4hZyqG6lFv9T3gZ7KyN4HvHdBlk9W+ySjfQqf9CmV+8R5UxC004WPfKlM7X6GoLqZxvSPwzCHTGLGtc+8JPoYaQ2PP07e/qVA1XeYNyPJZC0uxfvEueNmmUKw8vp+U1BLKRaU3KPCds7xDa6MoPotKrEC1Nz+asD7rIsIXhV5nfjjxX1SaXVPqylsLdTMxsfMFnzBe0+/Ym0g0SrlBGkWl69WKdrfeRATmeiFQpQQU0iuBWY7rA0r9eWMZXrNRqkUl37kGWO1sXl/S6tNbihCdzJjoIepyCCJWhcWv8t1+TKXGvtr4A4T45uuPAfxTlPXta5y84ET0hGgU/IgAPRkBfKlbRhBwmwI946Aqi+n0ZvTT8GG6XP5yI/gPdKzP1IpdRIkownCYegDrjVq/n2MQFtjb4WS4Hs/nqZ+XAg2e73PfjvL8aTbF8tMyEEgFAuPVHCSQGp0jekSWD4HAKjAe0EhOWf9lkZt6PfAQwJhhAQMOBD7c3Z5r5ixEmWkgNU/RYfj+94xFmdvJh1TAJan6mj1L0yUSFe7Go/UH8DJUukijxLKs2/g9BOrzc87dWOZ6/O2WxN4zsQ+IuoNVyla958WZ9MtTWOZxDaaxeLo+F/hwAhQdp0DxQL6fdd6Le4ptg2UW1+fUz4WtfcM/uysvCGuRNIzcOZUnR/VAoH7/GUCsL/7svniLVQs2UTsj7KBvk1Qg9dtsPamn6qM/uzPeYu5mOxOHtym3gYdP09niejgMPEyo7pChJuwZL1yNfp78soQYsni8n9cXlAgVxdFaW23OET7gAZP7xy/4J0ZYjgM3NBqluqeS3radN5Hb1McFLoADkOyX+R2VNlm/K1eulhhzczWFxuiLwCynWPHwjwIwqns6NWdq+x38FpXeWAR2AH+d36M0qAvNnUrU78ehKgCdu0JApVpAcW+bdd8WlcZ+DKHm9yi39WMIbXU5Z9f3BvIOFObt3LF1H83btH4mQlM9Npp6FzuiWWr3hrOb6dm5pvWUhR6rZd0/+t+p9EMISy2ld22/Z888OaUmFHd9u+4qpdFgWtO0DIV2vEW50rSC712gFA3M5nhWyPDVsNaKdWtBVHsbAAutGsISS/hpxbeJse75s57CYt5E80KziUIx3slmsx38T7ZUjGi5xIYmJGIRrSyJHBfi3mCfcPqy1JK4IJI4pPtpx6/lJlTUdquHhdlLk5IoNnrxSEGLRIqdbC8Wi7UqzXKzmc+XK72klkng7+UamS4Kchz8QcXTu7dHtC60zqGa1gD8Cns+lM7l0t7fblRqtWa1E2E23NUp0tf5MNrV/xc0L9KLF7RktSahLZE5MUxe1JuXjYiGwhyoXpvnUPLk3u3tQi4PuDVglgBEzftDMdyV+O7riUqkUA4F4c284peaacG2uKseL7DIZFkrxDiELJxi2XgyEkmWsr1Wt6zaUGpxxAN6Yb2A9qvHqRLpFZDIk14noXlv35pFTZ73UAylic1CQt1YK+07KajF86Aeke2X9yl0Ts5m2kiSAbleJOfxESWdK2jFltZBAB6HLpSSxIuUfd/V5RXiRWNSasFg3/NDBVS5EL2Ur53WQXGrBq4+0buqgXZwu3O0B744yiWRLKFa3NupmJIA3SPoNRgFtCPrp4cerbczGL0c/llDezxrBukRJHpElXikK3FS1V4s9dWVSqEZ5KTY9st72u2E0lWw2svczjuJZKek2V9JRgz0wOfFJC7ki1nsSjctg+VekX7qwaBn4L0C91ENeXgN5QL6JhJ78M+V03oOK9+BxKyEMmWEuh6f7uhaoYzCZohI5mz0CihPKe7aM1q6aoYsd1Pa41gjSOoWULiy+04xphuxjEYxpKdVZBLADpECapRRyCPMKRfZTAVICm+E2EKuH8TogU4rKo/thIZYITJX6tj+AIzYs/MleD0DhN1jopuVECeqinWHri9RsPJJlwd+Wq81Uc3D5yV0LYlCHGYCWNGUJAkbPAedq0qyt2M/WlgrCYpdIS7UNF5shsLl3Y8ql0iSekXkhWwB8aoqldy5/emSLLbCPH9gXNRQPyR7uDFFv+phD6BH0sSOOqbpApQdFPT6xkliLtQnry6CPDImsYTE4G7zGcjlSigdEiWPS3fDXF4V03TS3HIG7MAjtnvKZZjfJTggNa2F0euZv+Yi3RBmfEgBD+RpPyeJWVRSGhDTTRoO7AXtBrEkQmo62el4MmYYL8DXplMb1ToNSTxU+ZgQH7zweLmajWH0Stbvl5gnA3AXYChBiRJ69p75Rg7xsmGwaSTu5huZJkJVlhHhTd0DoUsJ4OOSdJSvm5P4wxMqb0WN6ET3bDcTibTlcJVFYi2HZJma7pnopUs9ySQKCVXeMYYc5GkheDGZjElelkKUTwejplHfS8cgsVIP/TROLndNheGBhHJSxfqVrWAim1YknpflPLVKlcVaYrgXxE0k+uFtgBohCF5ZbEwXORSseTRTlXBMa9JZeW1c6d7ezFMKnsmlhkSOC6q2A4hrTRmzCuBDvJqm9wgag6KlsxDqSIqUUXfUK470bpFEi0iyJXnVERw6SqFfShvGfjityCCvygnpkaOVSlkDf67gsYTkIFfa+fxJYt/fHMNz2GHwrO3ioygWKE3wG15uui0T+F7MNw+WYrEi76v1eEk7LOZ3woyGHZ/jEkvFZhD/WkLoso08yh3Hi8vQErFYmNQfOkh6wTFo4CC9BqYR5Qt6MMVTpLnHQeyTBBLl/nangcM7jo9hVVA+nlTaDqVDr4prFaxTRDjEpyG6Nl/gC+U9yqcHifLR6VmmlfWco/0CvQpuVU/AmHgXO40nQfmSCiTGRzW8V9jN4kW3J/GqUgLD7bzwpUKs43l9owh0cIL/isQ0UT5G+RjC50NoY+GiLG/UCrki4vtaCJmWldhnvLliNhaLvVz0Al+m2NugsSSrDZETm32gxTkm0eAR6nlGgPJV3mtgCVK+5Ta4qwLMtxF5hfgWW53dqgQrR9DhYdeSSLvmrp+AN5G6zq/JRlvmdbAsDJtSyiNU8Wgjna2Vm22dlxDaDC2FYsRhRoUKQlLORdCSKF8F1y9KqKfAhTEYnrWc3IX3wAhp4TiX4rKdMJIlCZKUl4asITAmfvsjxUY77Dub79vpJ0ghooJvUk1eVtAxpdkNHoleJd6riaSgL26MP4JfMsedqOKFEnBTzlJEC1VqPBdEzRy2Q6xJobyXdleKzZAHmTXihpsl5NSLbo3DS3Wt7T66lgxZqaDzfHj7ShdF5LsQV3QnIqAQlYqeV41WS8jw09uXLVW1BmhlNZJJZJKyW/3S5QvRGlyOkyA7yBOGZmmfUu7KnFzD+JKKTtCz+gOpnLat1eTbeVHEHXKYWpHvNUWkx7NIbm58NKuHpXzVumy2FPdgbBk1Jqk+lyAUVXStMzRQuBdU1T42OFAVfpcksEyuUkiGkVQyQU2756+k9XlzIBrKq3lVVwFMJ7XP6WWRl/FPGD0R1bw9T6MK+dquVrZQsKxCjyQru0yWI3qIBEJxMxw2OJF3qohpCOOcvMN1sl3VY46Ok/hGqTeNapW82iRLf8AswhbBd0mnkbkw0yqGOLOYKwu8xPkYgaSIwkjV9by+wT70ct4oTMBleaTucdssKkrStjHiCYH8ijB8s8NJSC6DxuWy5W7bCdTFSp93OchqrhbkjIUBt6jVkF9Sq4R4dzW8iaqqKuHeKTUE7AU89eYamFLRkgiVLd5QAoaIkP1uohdHYh7rWhLHnh4on+xWzp6kYtIR07DHkPbWOYoi0LGd0SoYvSTAx5vEXwOl4k12cyWFwnY/Mr2m7CofFRolr3pyQW2infWhI6WzkQT3kNpVdUwYEn0E7LqI0xE3+8qUM1Xk2EmJ6IKjQ8UCxBM86YBeh1F0Pq+72YeGAD25WkM1KYhD7z4pVy486tsVCUYLuFiLYplQRhQt02i4V3tiSeSULhmmlvFch+w1PcsnxwjwKNcKXRHJ7TKpfiVqOCUAv8htuORcM33hYrOEw4bazvuXZkkiSaJvSVT1sMs+FYlUYLMIiIv+wqJ7BrWk3XXRJApdMAnchNmBC60r2epDVmpNhe3kZF5yehVvZCVul6Mrqu57DSKLXMn3FQr3myqupAF62GVheNwTVCinVckp/yXRVgKQzjYkuY2nAaOXkPK66vbXFUzaxBq4r1D1pYDXyVeCO3oBdhpKE+bCh8iXk90IctY3M472ZbIdxDvLC2lUwIW3HXJeQJzfwJFwK3kCMgAgLflLoBcIszGF47mwy2Xk+mku5IyrSNBzLy/HCyounkWMZDmLNgwb/MQFVp343kVxS2ptVdwmfYoexMrVgBkIEY6voJwqOp0B+Pi88WM1g9zXLde2SEs6Tr6FS7Y+yzbuJtgLVMnk8zqYBsRcPJHxzdXjNEpfhJx+abvZZwXHRswdMaRKmOfcKGjAdSTsg2JhLlR5oVMZ1AyGtxdGwU4x5BdBa0pK1QhyReeIve6ZKxkmYEoWXZZFMe+0lFCJ8StBtMO0jxRQPslWvhgqM1kwN8R0zbwBez6XY+sXIGrYHy8YpFCuWC+wTK8TC0uNHJKIu0xuVsESgHwbKHPXoB4vGG+Sz+9UVLKIOK80slk8uqoFXdrTDVmKoOJ1YztsF5Bc3fR8lTCxFyXW8sv5GpKjPGnUVxiVy0vZnhlaI5thN3uRdRVmC0gMqVthF2kcJmayse7YFzdQKqMWzBYXBtUIch7bmRyp4u1Om5GlaM4b9hfGjF6hHArZmbqStZWvUoEe2JOekMV8U9wkFzXl9UL1IYVsCLuWkbBMupNgNKBlwQoyo1Yl5B5/EYHnt2dVg1Q3zm8kdBrSk4hX+0Fj6DgUOuMDjb4w0rS8jPO2HeN0Cd7dsFmQAndguOCWZLH4TizuUABg+MamA4xOs+Zau2tKohrkXBXROC4kvQ7OIfDhvSUWHsYaTh7naCJPIMG64pgmqCLP2Vs1ACdUwiV+J6ErIKkLuID+GTaBnbwTIS6RmsDOAL+fKXnSMEsSaLsWD+gZF2bzvPVWpdh1UgbNtpMGCkG2Jlnf7CEpFnKHtySSC7Q2kUO2a47evCPHcGem4WDbdZJL8palE8b+FBxZ7I7BEGuXeBXCKlvrokuHKkjMEH0kq6GEo+2NvZlt+Gz0GNeEqYWgk8TC6yJWMKWK1FbIYQugoPGOrbAMWX6I0LpHSDPie6GbM9dfzQ1pppvNuq7bQfgti0G3DEdDPmC6eODJxYsKWJ1o7qJUCAhGRxNlXPTDu9xg9HjdG+ZiN4G35AptxvOkjR6xfqkRaTRrF6FW2HENJEoVFU1HElZti86Bx+zimGJ78J4bvbRP1tINAcsAVegrZjUdOztn0YP4GcPDXyISNQ1VSzTNDrYc82xAfpwLqq5KXMaBL5MnPEgxdn0UrG/uy5eMZT/be8WQk5iWyZ4vIlKND1moXiEZ44c3y2ZICcToVAlJZYVkeYY9KRWyJG+i1/O3lx+TAOhABklpc/89cTpBex2uGzaVK13DRUWYYRlPtyaGRQNVAkLJABQ+0CP5lGUmV7YJFpEUwv8bhIV4A20zrOz0y3n3qo9s0pHJkhZqJXwTDb6fwazl5CSpViSY4sYLZgxOtJGEFYPAhycjo2LSaha+E7EM5y9jAzzCsbTqNEmKGY5SYFsToQNxczUWK10mU0WiFDQ+g41X1BPYRvAHSOHdJqJEh8B1poFEXhAXj3XbWr7g+f2rPQrPWy7kqous8JpI9okDsXwmiRYFo9vYftLFZMSIJE0gCMkriCGEoZBJBiKjdFBQKlibURPlq+CLK1WvC15uDhHmbcIH8+Ter0CyiiAP5Ny4oUipILx/QkItUz9JKAjKYJoXGXM8rvICrvJzSMc3DJHWsVLJVkTAerRVHnbEWM9A+bYMF6sQTDL49oRuHk+X9aky5ipaOlLeXu8ETxuGT4dRl7xMUstwXkJBXGUzLbdfDiGvVdDDRTFWGvqKhV4GBTdXZypIlmW4rE3OI21VrfUcn9U1PiDGjUGHUcU1DoiXoRDgXTO9WBx8U976btpWpKvdol+GIM/xcgjppD+5CkL5Tjq9sQdMaaMwhgmFdpKHSLev6l3Le5aNXubjrsXSmIwu3JhHWeHIx20owJBEvD5l7UwvoX5pk4yV2s1ma2MFkN3MteLOBxSIj+4CJ8tkOt1KpWGHgAtULjqMC5sUz0XKLd3DgWshIzy0DfDAgvtFxWSSrh5Gqs1mu+W9Qul+sdRsVmIRYxu5/X7CzV5YgR0xRz5xOtnt662E83Rz5fRFE8wZybqWQMT43+kbFpY0b79IAjOP5LDn/Q6KlowXNYO8g9nmDX8H5rJ798AxvdzL91j29vPJz131+2R9gTyW29iIYaHGGi9agh/jLQhRF6g9wjFFpL88acmajjpG/yJH3D2wJa/uHxduAsZzmv60ZxAK+NCJBciACPwQffEQihY2z35nL3okQLYrNXtZpQyZ8j6y86Ic8pCD6Wp6StOHCX6OFn4UvMu9CtHF4Pr662i5fnwc3t+N56l6/cOHD7/++mFD6iCfPn2af767u5tMhsPh4+N6vVwuR6Prm7/97QsBeXvKTbUVmEj50mEXV+gFqv2CmLf67gfvy+NytPwtMF4u1+vb49s/QjBkN6PpbPi0CtQBKgxWvR4IrMbjp/vhM0AznWJoduXbEsvj42w2HE4m49UqkAqkUh+MFlKr8RMAO11+G91cm4oLV2t0Wq66VTW8n2nvFdbWu33wCaxwMxnOhoFU/dfVgJbxbsRxQA2DNhljDfsOw53fDWfrJYwVBAwV/v16Mxot8fwBQDOM0bY8P8PL6/WUaN4NFvgSfPEGvohVeDh5Gs8DRIPhAuOnybAVKf/H366vAUvcB6BVZP3DeMwsec7sIS779buOBEZYDL/jx+z+tKDm+qJG7Fx8uVkP7+bYBOupVGA+nszW09ENgLVcz+6f7kCL6o6kDpK6W6Db8zEGa/ibYdQwBbPZx8nk9//UuL8bjQdWk/vH//rHf//zXyaUZNQQjg6gauzrbg8DGKjPH8bHUpeXGx2MZmMTkwAZ5ASsbvwZbK7+c/17ynrDn+Am3NDC3/l4NZ78T0dD/xxO7u/hkmP85O8UPvwAkLy7f15/+3o9WBw01oPOTRGYr7/+tLyvD2joHmlj8e0+9euvqTsww4lrrB7DN54KvwXIYcg97L5E/p3PU7+XNPR38th0cKvgVwHE+YMNNlFbsITHJbhK57GuRsjZMMCDTl8QmOfHm8Vy/o0CenCt258mk8fR9SB6uwR3NH9VfTxssv6hvke+f3d0zbPB+epzAKPH/W5fIWWAOp+vVq7eWNdMzSez5bevg9uoMfnC9nhe85FCNDUA1G/mNKrQguECFtfrMViMF3g2WNgbQhDBIXOJA8kXg/DhYEKiw/XAkS9fSIDBEWYNAWY4vH8aYzQ2gcctzgMPjQKS/v3atLmmwGzi83D59fYECxzcsgoMeuDrEZ0CdscYOuXLcjgmQzHk4SHgaBiMbzUePi6J//H/PGQBWMotxhuAJcxmeD95+lcOoX/Uv+OrGeGIsJzUp4DjJz7PgD+uh093c9wr80O4hx8Ckxm2HNyzKMl8/sBEYnFtxAuXhZHugcsGC8Gs7Nb22Tjw+T2gA5oQhO1iRzajS80FhtSkNHe///t//303J7Nnh3hgi8T3DW5B15ePw/H8E5lcQt/rgfFwPfqyOCzA+BUjXNysJ4R2fa+DTd5/XD+OV0/D2fTmC4S5jQGSLFcwGdjJAJrkTTAeDyywxl+WiUR6aDO3vUT/hzSiqOAZwAMQ0jQO1H/GKU59jpG6gXldYHq6XD8DmPfjwM94KKnP94DxwACRqKKPDhuy6x8FZjAargLzJzBKrGFA+vE1bv+YubO6Zd6zpVQKaHNrZATt7o/CSeTiFvRzOcO8FLLFD6kVNpAvC5J4sUD1v2KdTOHEJjCZAr7Wd2mcRrX5y+2tCZnZPPGCLK6D/XGPTTSfHNXB99O5lS+NeOTO1nDpJsq6ObOgAFaj6eP9CmeS9dU9CQLGB6K3g5vl81Mg9SE1nvwGmnhKbNkSYfDVdxtYaCJrPgqaVXNocz9rORSuHXwhMO5vUwh8gY+/fB24ixGgrNej9WyCjexx+XWQMKbgNOX49njKtzaE7vkB9o3mkVgJOduWGFwYlE/Y/R6FeLIe3n/8CZDa9EHgIJePQGzhjVMCH34IPnMfYFifD9O2ntJ28GX3X411FZViye7GnQ5JFFITh1/HjkHm76CNo+lshmP0rUGzjLREwCgCWb0+cmZIw7fPP/s+euvIk04G+8PQxvkB5Y65O9oQDUk6hVtPwXJB52azNVZGdy3p9sjoGF0+LZfT376PJ/fjqS/jO077HvfPF+s83gXfuVDjnbpeBKGy3/thBSf+4crbt/Xz8y/fro3a/AnU4ub31R3kYvXvSz/Ge9xTyAVh/lIx3LUW0eTwHRn2cjC6pOFft60EWy6klV8JoQV4owf7MWhoMYQM59PE14mSr5chNz892H/kCLsxFVm8/CxjlWMjfUT/mYGGCFEgZuD/vt4AitfHnIIHPb0Bpv596udYsKPjxnJc37eOsOlDE2SNXEbNJkKoSeeO+53uGNmOqfMsXrTBawSHHq8+GY+mk88+Uopjj+tg8THXo/103z0VBQnJoihK4PXo3I19mAjRxGKhHDKmxXy0iE5Tpx9Ac/z5YgPwF5MDHXUi3tR1vdnxecPPkYKJ1Xah1VtGa6y/o5OPbjSz01fhM3cbQIrFfHtOBeqLqPnrH5lKnygvOPYFfivKLE6kzQcfXcE6rH42BPic477O6jxBT3lhhMRpnpxyHB4yhMVwhVfPltPlFJc4V/eTyd3d3efV6EdQv7dok315zXlHph8C89VqNX56sArsgXrgnM8atuRtCk2vbXjY+rTADH4nS4wPn6wlivpsZ1PGGYrj2ikXlY6dluisvrEsd/NDeD5ziZPq+TNHs2UsAjua2yuS9cninI7w3SusvSmL2tFsB5wf4CUQom4/W8r36I7GZyu2ktA7lJKxGjqhsUXAge/8wbMKt5bxUkLv9GiEj7lOjbEJP5w/Y3EBdsgGj8ObPfFUAoEZpub12eP07pObOZ+rbJx4RPNQytOmQmAXgdT36fNg9HwPrIVKb95KtoMjtbjLHlff2+jSqD5ePkIEGXy8T6XO2Xq3U/nzOJRyMlkbR9NHH8epmzfJhujItnc6i5MBbwND46xhCLo3n4Y0u0QrLlrNMZsadxaHUn57jDJk4RADN7ijcU6zJTRPPzzdPb2tuFcoo+yCInyv3nJwTEv0GqMqGzVcXNKlGDxOJlO7DZ2Jqf6BQjWlOk/LfUs5oX6xtymrxvJ+hOZxzTTOi/qhhGVo+T3S2p96dPafcGHqA35fpntK4fsvMeXYXQ5/iVvoHaP97uS0FZe/xBD2XWYI1OS92S1tOnrcrtYfXugWld4bXTnkZvjDWzp1qfQk+X/c8qJgiqs/sgAAAABJRU5ErkJggg==)","metadata":{}},{"cell_type":"markdown","source":"Please leave a comment, if you have any suggestion on other feature selection techniques that are beneficial. I will continue to improve this notebook for knowledge sharing.","metadata":{}}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}