{"cells":[{"metadata":{"_uuid":"4109f0ed3489ea12536e0e8048d83b04e3fc3eb5"},"cell_type":"markdown","source":"# **Feature Selector Baseline Don't Overfit**\n---\n![](https://static1.squarespace.com/static/5213a664e4b01a5565dc90f1/t/5bc4e0c4e4966bc550291202/1546736393368/Machine+Learning+Generalization)\n\n\n## ***Outline of the Notebook***\n\n---\n\n* [**Step.1 Read Dataset**](#Step.1-Read-Dataset)\n* [**Step.2 Display dataset**](#Step.2-Display-dataset)\n* [**Step.3 Remove unwanted columns**](#Step.3-Remove-unwanted-columns)\n* [**Step.4 Create Instance of Feature Selector**](#Step.4-Create-Instance-of-Feature-Selector)\n* [**Step.5 Missing Value**](#Step.5-Missing-Value)\n* [**Step.6 Single Unique Value**](#Step.6-Single-Unique-Value)\n* [**Step.7 Plot Feature Importances**](#Step.7-Plot-Feature-Importances)\n* [**Step.8 Low Importance Features**](#Step.8-Low-Importance-Features)\n* [**Step.9 Removing Features**](#Step.9-Removing-Features)\n* [**Step.10 Handling One-Hot Features**](#Step.10-Handling-One-Hot Features)\n* [**Step.11 Model Training**](#Step.11-Model-Training)\n* [**Step.12 Model Evaluation Framework to check Importance**](#Step.12-Model-Evaluation-Framework-to-check Importance)\n* [**Step.13 Model Training and Evaluation Framework to check Importance**](#Step.13-Model-Training-and-Evaluation-Framework-to-check-Importance)\n\n---\n**Reference Github : https://github.com/WillKoehrsen/feature-selector**  \n**Reference Kernel : **  \n**1. https://www.kaggle.com/sovchinnikov/logistic-regression**  \n**2. https://www.kaggle.com/artgor/how-to-not-overfit/**\n"},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport time\nimport lightgbm as lgb\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import KFold\n# Data processing, metrics and modeling\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import GridSearchCV, cross_val_score\nfrom sklearn.feature_selection import RFE\n\n# load libraries\nfrom sklearn import model_selection\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn import datasets\nfrom sklearn.model_selection import train_test_split\n\n#ignore warning messages \nimport warnings\nwarnings.filterwarnings('ignore') \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport eli5\nfrom eli5.sklearn import PermutationImportance\nimport shap","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"588ad66de581dfc62d307ee07512e38cd384bc93","trusted":true},"cell_type":"code","source":"# numpy and pandas for data manipulation\nimport pandas as pd\nimport numpy as np\n\n# model used for feature importances\nimport lightgbm as lgb\n\n# utility for early stopping with a validation set\nfrom sklearn.model_selection import train_test_split\n\n# visualizations\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# memory management\nimport gc\n\n# utilities\nfrom itertools import chain\n\nclass FeatureSelector():\n    \"\"\"\n    Class for performing feature selection for machine learning or data preprocessing.\n    \n    Implements five different methods to identify features for removal \n    \n        1. Find columns with a missing percentage greater than a specified threshold\n        2. Find columns with a single unique value\n        3. Find collinear variables with a correlation greater than a specified correlation coefficient\n        4. Find features with 0.0 feature importance from a gradient boosting machine (gbm)\n        5. Find low importance features that do not contribute to a specified cumulative feature importance from the gbm\n        \n    Parameters\n    --------\n        data : dataframe\n            A dataset with observations in the rows and features in the columns\n        labels : array or series, default = None\n            Array of labels for training the machine learning model to find feature importances. These can be either binary labels\n            (if task is 'classification') or continuous targets (if task is 'regression').\n            If no labels are provided, then the feature importance based methods are not available.\n        \n    Attributes\n    --------\n    \n    ops : dict\n        Dictionary of operations run and features identified for removal\n        \n    missing_stats : dataframe\n        The fraction of missing values for all features\n    \n    record_missing : dataframe\n        The fraction of missing values for features with missing fraction above threshold\n        \n    unique_stats : dataframe\n        Number of unique values for all features\n    \n    record_single_unique : dataframe\n        Records the features that have a single unique value\n        \n    corr_matrix : dataframe\n        All correlations between all features in the data\n    \n    record_collinear : dataframe\n        Records the pairs of collinear variables with a correlation coefficient above the threshold\n        \n    feature_importances : dataframe\n        All feature importances from the gradient boosting machine\n    \n    record_zero_importance : dataframe\n        Records the zero importance features in the data according to the gbm\n    \n    record_low_importance : dataframe\n        Records the lowest importance features not needed to reach the threshold of cumulative importance according to the gbm\n    \n    \n    Notes\n    --------\n    \n        - All 5 operations can be run with the `identify_all` method.\n        - If using feature importances, one-hot encoding is used for categorical variables which creates new columns\n    \n    \"\"\"\n    \n    def __init__(self, data, labels=None):\n        \n        # Dataset and optional training labels\n        self.data = data\n        self.labels = labels\n\n        if labels is None:\n            print('No labels provided. Feature importance based methods are not available.')\n        \n        self.base_features = list(data.columns)\n        self.one_hot_features = None\n        \n        # Dataframes recording information about features to remove\n        self.record_missing = None\n        self.record_single_unique = None\n        self.record_collinear = None\n        self.record_zero_importance = None\n        self.record_low_importance = None\n        \n        self.missing_stats = None\n        self.unique_stats = None\n        self.corr_matrix = None\n        self.feature_importances = None\n        \n        # Dictionary to hold removal operations\n        self.ops = {}\n        \n        self.one_hot_correlated = False\n        \n    def identify_missing(self, missing_threshold):\n        \"\"\"Find the features with a fraction of missing values above `missing_threshold`\"\"\"\n        \n        self.missing_threshold = missing_threshold\n\n        # Calculate the fraction of missing in each column \n        missing_series = self.data.isnull().sum() / self.data.shape[0]\n        self.missing_stats = pd.DataFrame(missing_series).rename(columns = {'index': 'feature', 0: 'missing_fraction'})\n\n        # Sort with highest number of missing values on top\n        self.missing_stats = self.missing_stats.sort_values('missing_fraction', ascending = False)\n\n        # Find the columns with a missing percentage above the threshold\n        record_missing = pd.DataFrame(missing_series[missing_series > missing_threshold]).reset_index().rename(columns = \n                                                                                                               {'index': 'feature', \n                                                                                                                0: 'missing_fraction'})\n\n        to_drop = list(record_missing['feature'])\n\n        self.record_missing = record_missing\n        self.ops['missing'] = to_drop\n        \n        print('%d features with greater than %0.2f missing values.\\n' % (len(self.ops['missing']), self.missing_threshold))\n        \n    def identify_single_unique(self):\n        \"\"\"Finds features with only a single unique value. NaNs do not count as a unique value. \"\"\"\n\n        # Calculate the unique counts in each column\n        unique_counts = self.data.nunique()\n        self.unique_stats = pd.DataFrame(unique_counts).rename(columns = {'index': 'feature', 0: 'nunique'})\n        self.unique_stats = self.unique_stats.sort_values('nunique', ascending = True)\n        \n        # Find the columns with only one unique count\n        record_single_unique = pd.DataFrame(unique_counts[unique_counts == 1]).reset_index().rename(columns = {'index': 'feature', \n                                                                                                                0: 'nunique'})\n\n        to_drop = list(record_single_unique['feature'])\n    \n        self.record_single_unique = record_single_unique\n        self.ops['single_unique'] = to_drop\n        \n        print('%d features with a single unique value.\\n' % len(self.ops['single_unique']))\n    \n    def identify_collinear(self, correlation_threshold, one_hot=False):\n        \"\"\"\n        Finds collinear features based on the correlation coefficient between features. \n        For each pair of features with a correlation coefficient greather than `correlation_threshold`,\n        only one of the pair is identified for removal. \n        Using code adapted from: https://chrisalbon.com/machine_learning/feature_selection/drop_highly_correlated_features/\n        \n        Parameters\n        --------\n        correlation_threshold : float between 0 and 1\n            Value of the Pearson correlation cofficient for identifying correlation features\n        one_hot : boolean, default = False\n            Whether to one-hot encode the features before calculating the correlation coefficients\n        \"\"\"\n        \n        self.correlation_threshold = correlation_threshold\n        self.one_hot_correlated = one_hot\n        \n         # Calculate the correlations between every column\n        if one_hot:\n            \n            # One hot encoding\n            features = pd.get_dummies(self.data)\n            self.one_hot_features = [column for column in features.columns if column not in self.base_features]\n\n            # Add one hot encoded data to original data\n            self.data_all = pd.concat([features[self.one_hot_features], self.data], axis = 1)\n            \n            corr_matrix = pd.get_dummies(features).corr()\n\n        else:\n            corr_matrix = self.data.corr()\n        \n        self.corr_matrix = corr_matrix\n    \n        # Extract the upper triangle of the correlation matrix\n        upper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k = 1).astype(np.bool))\n        \n        # Select the features with correlations above the threshold\n        # Need to use the absolute value\n        to_drop = [column for column in upper.columns if any(upper[column].abs() > correlation_threshold)]\n\n        # Dataframe to hold correlated pairs\n        record_collinear = pd.DataFrame(columns = ['drop_feature', 'corr_feature', 'corr_value'])\n\n        # Iterate through the columns to drop to record pairs of correlated features\n        for column in to_drop:\n\n            # Find the correlated features\n            corr_features = list(upper.index[upper[column].abs() > correlation_threshold])\n\n            # Find the correlated values\n            corr_values = list(upper[column][upper[column].abs() > correlation_threshold])\n            drop_features = [column for _ in range(len(corr_features))]    \n\n            # Record the information (need a temp df for now)\n            temp_df = pd.DataFrame.from_dict({'drop_feature': drop_features,\n                                             'corr_feature': corr_features,\n                                             'corr_value': corr_values})\n\n            # Add to dataframe\n            record_collinear = record_collinear.append(temp_df, ignore_index = True)\n\n        self.record_collinear = record_collinear\n        self.ops['collinear'] = to_drop\n        \n        print('%d features with a correlation magnitude greater than %0.2f.\\n' % (len(self.ops['collinear']), self.correlation_threshold))\n\n    def identify_zero_importance(self, task, eval_metric=None, \n                                 n_iterations=10, early_stopping = True):\n        \"\"\"\n        \n        Identify the features with zero importance according to a gradient boosting machine.\n        The gbm can be trained with early stopping using a validation set to prevent overfitting. \n        The feature importances are averaged over `n_iterations` to reduce variance. \n        \n        Uses the LightGBM implementation (http://lightgbm.readthedocs.io/en/latest/index.html)\n        Parameters \n        --------\n        eval_metric : string\n            Evaluation metric to use for the gradient boosting machine for early stopping. Must be\n            provided if `early_stopping` is True\n        task : string\n            The machine learning task, either 'classification' or 'regression'\n        n_iterations : int, default = 10\n            Number of iterations to train the gradient boosting machine\n            \n        early_stopping : boolean, default = True\n            Whether or not to use early stopping with a validation set when training\n        \n        \n        Notes\n        --------\n        \n        - Features are one-hot encoded to handle the categorical variables before training.\n        - The gbm is not optimized for any particular task and might need some hyperparameter tuning\n        - Feature importances, including zero importance features, can change across runs\n        \"\"\"\n\n        if early_stopping and eval_metric is None:\n            raise ValueError(\"\"\"eval metric must be provided with early stopping. Examples include \"auc\" for classification or\n                             \"l2\" for regression.\"\"\")\n            \n        if self.labels is None:\n            raise ValueError(\"No training labels provided.\")\n        \n        # One hot encoding\n        features = pd.get_dummies(self.data)\n        self.one_hot_features = [column for column in features.columns if column not in self.base_features]\n\n        # Add one hot encoded data to original data\n        self.data_all = pd.concat([features[self.one_hot_features], self.data], axis = 1)\n\n        # Extract feature names\n        feature_names = list(features.columns)\n\n        # Convert to np array\n        features = np.array(features)\n        labels = np.array(self.labels).reshape((-1, ))\n\n        # Empty array for feature importances\n        feature_importance_values = np.zeros(len(feature_names))\n        \n        print('Training Gradient Boosting Model\\n')\n        \n        # Iterate through each fold\n        for _ in range(n_iterations):\n\n            if task == 'classification':\n                model = lgb.LGBMClassifier(n_estimators=1000, learning_rate = 0.05, verbose = -1)\n\n            elif task == 'regression':\n                model = lgb.LGBMRegressor(n_estimators=1000, learning_rate = 0.05, verbose = -1)\n\n            else:\n                raise ValueError('Task must be either \"classification\" or \"regression\"')\n                \n            # If training using early stopping need a validation set\n            if early_stopping:\n                \n                train_features, valid_features, train_labels, valid_labels = train_test_split(features, labels, test_size = 0.15)\n\n                # Train the model with early stopping\n                model.fit(train_features, train_labels, eval_metric = eval_metric,\n                          eval_set = [(valid_features, valid_labels)],\n                          early_stopping_rounds = 100, verbose = -1)\n                \n                # Clean up memory\n                gc.enable()\n                del train_features, train_labels, valid_features, valid_labels\n                gc.collect()\n                \n            else:\n                model.fit(features, labels)\n\n            # Record the feature importances\n            feature_importance_values += model.feature_importances_ / n_iterations\n\n        feature_importances = pd.DataFrame({'feature': feature_names, 'importance': feature_importance_values})\n\n        # Sort features according to importance\n        feature_importances = feature_importances.sort_values('importance', ascending = False).reset_index(drop = True)\n\n        # Normalize the feature importances to add up to one\n        feature_importances['normalized_importance'] = feature_importances['importance'] / feature_importances['importance'].sum()\n        feature_importances['cumulative_importance'] = np.cumsum(feature_importances['normalized_importance'])\n\n        # Extract the features with zero importance\n        record_zero_importance = feature_importances[feature_importances['importance'] == 0.0]\n        \n        to_drop = list(record_zero_importance['feature'])\n\n        self.feature_importances = feature_importances\n        self.record_zero_importance = record_zero_importance\n        self.ops['zero_importance'] = to_drop\n        \n        print('\\n%d features with zero importance after one-hot encoding.\\n' % len(self.ops['zero_importance']))\n    \n    def identify_low_importance(self, cumulative_importance):\n        \"\"\"\n        Finds the lowest importance features not needed to account for `cumulative_importance` fraction\n        of the total feature importance from the gradient boosting machine. As an example, if cumulative\n        importance is set to 0.95, this will retain only the most important features needed to \n        reach 95% of the total feature importance. The identified features are those not needed.\n        Parameters\n        --------\n        cumulative_importance : float between 0 and 1\n            The fraction of cumulative importance to account for \n        \"\"\"\n\n        self.cumulative_importance = cumulative_importance\n        \n        # The feature importances need to be calculated before running\n        if self.feature_importances is None:\n            raise NotImplementedError(\"\"\"Feature importances have not yet been determined. \n                                         Call the `identify_zero_importance` method first.\"\"\")\n            \n        # Make sure most important features are on top\n        self.feature_importances = self.feature_importances.sort_values('cumulative_importance')\n\n        # Identify the features not needed to reach the cumulative_importance\n        record_low_importance = self.feature_importances[self.feature_importances['cumulative_importance'] > cumulative_importance]\n\n        to_drop = list(record_low_importance['feature'])\n\n        self.record_low_importance = record_low_importance\n        self.ops['low_importance'] = to_drop\n    \n        print('%d features required for cumulative importance of %0.2f after one hot encoding.' % (len(self.feature_importances) -\n                                                                            len(self.record_low_importance), self.cumulative_importance))\n        print('%d features do not contribute to cumulative importance of %0.2f.\\n' % (len(self.ops['low_importance']),\n                                                                                               self.cumulative_importance))\n        \n    def identify_all(self, selection_params):\n        \"\"\"\n        Use all five of the methods to identify features to remove.\n        \n        Parameters\n        --------\n            \n        selection_params : dict\n           Parameters to use in the five feature selection methhods.\n           Params must contain the keys ['missing_threshold', 'correlation_threshold', 'eval_metric', 'task', 'cumulative_importance']\n        \n        \"\"\"\n        \n        # Check for all required parameters\n        for param in ['missing_threshold', 'correlation_threshold', 'eval_metric', 'task', 'cumulative_importance']:\n            if param not in selection_params.keys():\n                raise ValueError('%s is a required parameter for this method.' % param)\n        \n        # Implement each of the five methods\n        self.identify_missing(selection_params['missing_threshold'])\n        self.identify_single_unique()\n        self.identify_collinear(selection_params['correlation_threshold'])\n        self.identify_zero_importance(task = selection_params['task'], eval_metric = selection_params['eval_metric'])\n        self.identify_low_importance(selection_params['cumulative_importance'])\n        \n        # Find the number of features identified to drop\n        self.all_identified = set(list(chain(*list(self.ops.values()))))\n        self.n_identified = len(self.all_identified)\n        \n        print('%d total features out of %d identified for removal after one-hot encoding.\\n' % (self.n_identified, \n                                                                                                  self.data_all.shape[1]))\n        \n    def check_removal(self, keep_one_hot=True):\n        \n        \"\"\"Check the identified features before removal. Returns a list of the unique features identified.\"\"\"\n        \n        self.all_identified = set(list(chain(*list(self.ops.values()))))\n        print('Total of %d features identified for removal' % len(self.all_identified))\n        \n        if not keep_one_hot:\n            if self.one_hot_features is None:\n                print('Data has not been one-hot encoded')\n            else:\n                one_hot_to_remove = [x for x in self.one_hot_features if x not in self.all_identified]\n                print('%d additional one-hot features can be removed' % len(one_hot_to_remove))\n        \n        return list(self.all_identified)\n        \n    \n    def remove(self, methods, keep_one_hot = True):\n        \"\"\"\n        Remove the features from the data according to the specified methods.\n        \n        Parameters\n        --------\n            methods : 'all' or list of methods\n                If methods == 'all', any methods that have identified features will be used\n                Otherwise, only the specified methods will be used.\n                Can be one of ['missing', 'single_unique', 'collinear', 'zero_importance', 'low_importance']\n            keep_one_hot : boolean, default = True\n                Whether or not to keep one-hot encoded features\n                \n        Return\n        --------\n            data : dataframe\n                Dataframe with identified features removed\n                \n        \n        Notes \n        --------\n            - If feature importances are used, the one-hot encoded columns will be added to the data (and then may be removed)\n            - Check the features that will be removed before transforming data!\n        \n        \"\"\"\n        \n        \n        features_to_drop = []\n      \n        if methods == 'all':\n            \n            # Need to use one-hot encoded data as well\n            data = self.data_all\n                                          \n            print('{} methods have been run\\n'.format(list(self.ops.keys())))\n            \n            # Find the unique features to drop\n            features_to_drop = set(list(chain(*list(self.ops.values()))))\n            \n        else:\n            # Need to use one-hot encoded data as well\n            if 'zero_importance' in methods or 'low_importance' in methods or self.one_hot_correlated:\n                data = self.data_all\n                \n            else:\n                data = self.data\n                \n            # Iterate through the specified methods\n            for method in methods:\n                \n                # Check to make sure the method has been run\n                if method not in self.ops.keys():\n                    raise NotImplementedError('%s method has not been run' % method)\n                    \n                # Append the features identified for removal\n                else:\n                    features_to_drop.append(self.ops[method])\n        \n            # Find the unique features to drop\n            features_to_drop = set(list(chain(*features_to_drop)))\n            \n        features_to_drop = list(features_to_drop)\n            \n        if not keep_one_hot:\n            \n            if self.one_hot_features is None:\n                print('Data has not been one-hot encoded')\n            else:\n                             \n                features_to_drop = list(set(features_to_drop) | set(self.one_hot_features))\n       \n        # Remove the features and return the data\n        data = data.drop(columns = features_to_drop)\n        self.removed_features = features_to_drop\n        \n        if not keep_one_hot:\n        \tprint('Removed %d features including one-hot features.' % len(features_to_drop))\n        else:\n        \tprint('Removed %d features.' % len(features_to_drop))\n        \n        return data\n    \n    def plot_missing(self):\n        \"\"\"Histogram of missing fraction in each feature\"\"\"\n        if self.record_missing is None:\n            raise NotImplementedError(\"Missing values have not been calculated. Run `identify_missing`\")\n        \n        self.reset_plot()\n        \n        # Histogram of missing values\n        plt.style.use('seaborn-white')\n        plt.figure(figsize = (7, 5))\n        plt.hist(self.missing_stats['missing_fraction'], bins = np.linspace(0, 1, 11), edgecolor = 'k', color = 'red', linewidth = 1.5)\n        plt.xticks(np.linspace(0, 1, 11));\n        plt.xlabel('Missing Fraction', size = 14); plt.ylabel('Count of Features', size = 14); \n        plt.title(\"Fraction of Missing Values Histogram\", size = 16);\n        \n    \n    def plot_unique(self):\n        \"\"\"Histogram of number of unique values in each feature\"\"\"\n        if self.record_single_unique is None:\n            raise NotImplementedError('Unique values have not been calculated. Run `identify_single_unique`')\n        \n        self.reset_plot()\n\n        # Histogram of number of unique values\n        self.unique_stats.plot.hist(edgecolor = 'k', figsize = (7, 5))\n        plt.ylabel('Frequency', size = 14); plt.xlabel('Unique Values', size = 14); \n        plt.title('Number of Unique Values Histogram', size = 16);\n        \n    \n    def plot_collinear(self, plot_all = False):\n        \"\"\"\n        Heatmap of the correlation values. If plot_all = True plots all the correlations otherwise\n        plots only those features that have a correlation above the threshold\n        \n        Notes\n        --------\n            - Not all of the plotted correlations are above the threshold because this plots\n            all the variables that have been idenfitied as having even one correlation above the threshold\n            - The features on the x-axis are those that will be removed. The features on the y-axis\n            are the correlated features with those on the x-axis\n        \n        Code adapted from https://seaborn.pydata.org/examples/many_pairwise_correlations.html\n        \"\"\"\n        \n        if self.record_collinear is None:\n            raise NotImplementedError('Collinear features have not been idenfitied. Run `identify_collinear`.')\n        \n        if plot_all:\n        \tcorr_matrix_plot = self.corr_matrix\n        \ttitle = 'All Correlations'\n        \n        else:\n\t        # Identify the correlations that were above the threshold\n\t        # columns (x-axis) are features to drop and rows (y_axis) are correlated pairs\n\t        corr_matrix_plot = self.corr_matrix.loc[list(set(self.record_collinear['corr_feature'])), \n\t                                                list(set(self.record_collinear['drop_feature']))]\n\n\t        title = \"Correlations Above Threshold\"\n\n       \n        f, ax = plt.subplots(figsize=(10, 8))\n        \n        # Diverging colormap\n        cmap = sns.diverging_palette(220, 10, as_cmap=True)\n\n        # Draw the heatmap with a color bar\n        sns.heatmap(corr_matrix_plot, cmap=cmap, center=0,\n                    linewidths=.25, cbar_kws={\"shrink\": 0.6})\n\n        # Set the ylabels \n        ax.set_yticks([x + 0.5 for x in list(range(corr_matrix_plot.shape[0]))])\n        ax.set_yticklabels(list(corr_matrix_plot.index), size = int(160 / corr_matrix_plot.shape[0]));\n\n        # Set the xlabels \n        ax.set_xticks([x + 0.5 for x in list(range(corr_matrix_plot.shape[1]))])\n        ax.set_xticklabels(list(corr_matrix_plot.columns), size = int(160 / corr_matrix_plot.shape[1]));\n        plt.title(title, size = 14)\n        \n    def plot_feature_importances(self, plot_n = 15, threshold = None):\n        \"\"\"\n        Plots `plot_n` most important features and the cumulative importance of features.\n        If `threshold` is provided, prints the number of features needed to reach `threshold` cumulative importance.\n        Parameters\n        --------\n        \n        plot_n : int, default = 15\n            Number of most important features to plot. Defaults to 15 or the maximum number of features whichever is smaller\n        \n        threshold : float, between 0 and 1 default = None\n            Threshold for printing information about cumulative importances\n        \"\"\"\n        \n        if self.record_zero_importance is None:\n            raise NotImplementedError('Feature importances have not been determined. Run `idenfity_zero_importance`')\n            \n        # Need to adjust number of features if greater than the features in the data\n        if plot_n > self.feature_importances.shape[0]:\n            plot_n = self.feature_importances.shape[0] - 1\n\n        self.reset_plot()\n        \n        # Make a horizontal bar chart of feature importances\n        plt.figure(figsize = (10, 6))\n        ax = plt.subplot()\n\n        # Need to reverse the index to plot most important on top\n        # There might be a more efficient method to accomplish this\n        ax.barh(list(reversed(list(self.feature_importances.index[:plot_n]))), \n                self.feature_importances['normalized_importance'][:plot_n], \n                align = 'center', edgecolor = 'k')\n\n        # Set the yticks and labels\n        ax.set_yticks(list(reversed(list(self.feature_importances.index[:plot_n]))))\n        ax.set_yticklabels(self.feature_importances['feature'][:plot_n], size = 12)\n\n        # Plot labeling\n        plt.xlabel('Normalized Importance', size = 16); plt.title('Feature Importances', size = 18)\n        plt.show()\n\n        # Cumulative importance plot\n        plt.figure(figsize = (6, 4))\n        plt.plot(list(range(1, len(self.feature_importances) + 1)), self.feature_importances['cumulative_importance'], 'r-')\n        plt.xlabel('Number of Features', size = 14); plt.ylabel('Cumulative Importance', size = 14); \n        plt.title('Cumulative Feature Importance', size = 16);\n\n        if threshold:\n\n            # Index of minimum number of features needed for cumulative importance threshold\n            # np.where returns the index so need to add 1 to have correct number\n            importance_index = np.min(np.where(self.feature_importances['cumulative_importance'] > threshold))\n            plt.vlines(x = importance_index + 1, ymin = 0, ymax = 1, linestyles='--', colors = 'blue')\n            plt.show();\n\n            print('%d features required for %0.2f of cumulative importance' % (importance_index + 1, threshold))\n\n    def reset_plot(self):\n        plt.rcParams = plt.rcParamsDefault","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"23af79c91a83d141466d00fd22b42a8223beb0f4"},"cell_type":"markdown","source":"## **Step.1 Read Dataset**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a8ba6eb8e18e0f60b72613070ed02bcf5f16fa84"},"cell_type":"markdown","source":"## **Step.2 Display dataset**"},{"metadata":{"_uuid":"9dcbe24cda9c132b288c9f9dd6b9010e96357f3c","trusted":true},"cell_type":"code","source":"train.head(7)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ce3e9af7e6df877525acd037368d6dcc97eaefe"},"cell_type":"markdown","source":"## **Step.3 Remove unwanted columns**"},{"metadata":{"_uuid":"ed398fb5c07b7b1e2a70a828abc3756e229e6c0e","trusted":true},"cell_type":"code","source":"train.set_index('id')\ntest.set_index('id')\nprint(\"Done..\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d3b357ade7a994add095aaf5efad50d5c28e3987","trusted":true},"cell_type":"code","source":"labels = train.pop('target')\ntrain = train","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8a8a804afc79b329cd04c64d233e0989dfbe97e0"},"cell_type":"markdown","source":"## **Step.4 Create Instance of Feature Selector**"},{"metadata":{"_uuid":"7ef2c0cd8b3f736542806ffde4ee9e0e79da1d48","trusted":true},"cell_type":"code","source":"fs = FeatureSelector(data = train, labels = labels)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"25970e1fd6f980c0f2b442586197dab0e927a7a4"},"cell_type":"markdown","source":"## **Step.5 Missing Value**\n\nThe first feature selection method is straightforward: find any columns with a missing fraction greater than a specified threshold. For this example we will use a threhold of 0.6 which corresponds to finding features with more than 60% missing values. (This method does not one-hot encode the features first)."},{"metadata":{"_uuid":"a7737afcca915f7bc3eed6299c5163a741bbdeea","trusted":true},"cell_type":"code","source":"fs.identify_missing(missing_threshold=0.6)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"922bc051ab5c6a55d654e36a19fd44b2d8bbda48"},"cell_type":"markdown","source":"The features identified for removal can be accessed through the ops dictionary of the FeatureSelector object."},{"metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"_uuid":"b8cf71d0bfba11fb3d48ed9c4d333e2da6c7bd30","trusted":true},"cell_type":"code","source":"missing_features = fs.ops['missing']\nmissing_features[:10]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a1504533e69f03aeb193db99e014850f16237055","trusted":true},"cell_type":"code","source":"fs.plot_missing()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e57e228bd89cf76670d041c74340690df87ac990","trusted":true},"cell_type":"code","source":"fs.missing_stats.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f5f9996ed9a5ed98d564161860886bbda018cb65"},"cell_type":"markdown","source":"## **Step.6 Single Unique Value**"},{"metadata":{"_uuid":"516b0f52503e541e0acf906c93b4d0a4e44b2103","trusted":true},"cell_type":"code","source":"fs.identify_single_unique()\nsingle_unique = fs.ops['single_unique']\nfs.plot_unique()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"02fa1da3d1867eae80f0c21fe7b86e9a0984d0d0","trusted":true},"cell_type":"code","source":"fs.unique_stats.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5ea2b396f26ee403482b22ccb3e4424c271aa11e","trusted":true},"cell_type":"code","source":"fs.identify_collinear(correlation_threshold=0.5)\ncorrelated_features = fs.ops['collinear']\ncorrelated_features[:5]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"606bc799893cee781bab0d6cb8827d46f77e3277"},"cell_type":"markdown","source":"we have no single co-linear feature as per threshold"},{"metadata":{"_uuid":"0fd998034ddbf9e782cbe90e43ef30eda0cdfa3e","trusted":true},"cell_type":"code","source":"fs.plot_collinear(plot_all=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"03fe28d230d44bfce6715458a882224934acd04c","trusted":true},"cell_type":"code","source":"fs.record_collinear.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b75aecfe9204e56baa5c6df875ce52e6809e803b","trusted":true},"cell_type":"code","source":"fs.identify_zero_importance(task = 'classification', eval_metric = 'auc', n_iterations = 10, early_stopping = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dde99d6cb14ba9fbc081a30402650a1ed556034f"},"cell_type":"markdown","source":"Running the gradient boosting model requires one hot encoding the features. These features are saved in the one_hot_features attribute of the FeatureSelector. The original features are saved in the base_features."},{"metadata":{"_uuid":"8fcdf215e35473cbd5d7b9e9f86c5f07f4112338","trusted":true},"cell_type":"code","source":"one_hot_features = fs.one_hot_features\nbase_features = fs.base_features\nprint('There are %d original features' % len(base_features))\nprint('There are %d one-hot features' % len(one_hot_features))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d094a383be38fe7edd7169d387377e8f0822f9ee","trusted":true},"cell_type":"code","source":"fs.data_all.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"64523ea01c7458fa24c3b26474c3148d5207be03","trusted":true},"cell_type":"code","source":"zero_importance_features = fs.ops['zero_importance']\nzero_importance_features[10:15]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5caff6fb0cee2bd28ee1ddbd98b6e101f489ec59"},"cell_type":"markdown","source":"## **Step.7 Plot Feature Importances**\n\nThe feature importance plot using plot_feature_importances will show us the plot_n most important features (on a normalized scale where the features sum to 1). It also shows us the cumulative feature importance versus the number of features.\n\nWhen we plot the feature importances, we can pass in a threshold which identifies the number of features required to reach a specified cumulative feature importance. For example, threshold = 0.99 will tell us the number of features needed to account for 99% of the total importance."},{"metadata":{"_uuid":"fb31fb2fcd2b2725a19e4c3061fe54c186554eac","trusted":true},"cell_type":"code","source":"fs.plot_feature_importances(threshold = 0.99, plot_n = 12)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"813e6f302463e708f1b9e38db57e7201c8902b53","trusted":true},"cell_type":"code","source":"fs.feature_importances.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ea26980e6804b05b662d4c973f6b62815ed8a15b"},"cell_type":"markdown","source":"We could use these results to select only the 'n' most important features. For example, if we want the top 100 most importance, we could do the following."},{"metadata":{"_uuid":"1ab53da0db7534cc1b956f2abaa4a1cd7be51c4e","trusted":true},"cell_type":"code","source":"one_hundred_features = list(fs.feature_importances.loc[:99, 'feature'])\nlen(one_hundred_features)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6dba7bbe9c472fb626c9cccc40270c61c8d9df7d"},"cell_type":"markdown","source":"## **Step.8 Low Importance Features**\n\nThis method builds off the feature importances from the gradient boosting machine (identify_zero_importance must be run first) by finding the lowest importance features not needed to reach a specified cumulative total feature importance. For example, if we pass in 0.99, this will find the lowest important features that are not needed to reach 99% of the total feature importance.\n\nWhen using this method, we must have already run identify_zero_importance and need to pass in a cumulative_importance that accounts for that fraction of total feature importance.\n\n**Note of caution:** this method builds on the gradient boosting model features importances and again is non-deterministic. I advise running these two methods several times with varying parameters and testing each resulting set of features rather than picking one number and sticking to it."},{"metadata":{"_uuid":"b684842bca92b1d834777f51c07771ec2374bd6c","trusted":true},"cell_type":"code","source":"fs.identify_low_importance(cumulative_importance = 0.99)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"976586d73997fb15af1755b693e36afe6280685c"},"cell_type":"markdown","source":"The low importance features to remove are those that do not contribute to the specified cumulative importance. These are also available in the ops dictionary."},{"metadata":{"_uuid":"ea46b59b4293eb59ec4cee4b927ad00c61d94ad0","trusted":true},"cell_type":"code","source":"low_importance_features = fs.ops['low_importance']\nlow_importance_features[:5]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"79ab8f9fe10fe7d8c81d9bd1238cefa7e70fdff8"},"cell_type":"markdown","source":"## **Step.9 Removing Features**\n\nOnce we have identified the features to remove, we have a number of ways to drop the features. We can access any of the feature lists in the removal_ops dictionary and remove the columns manually. We also can use the remove method, passing in the methods that identified the features we want to remove.\n\nThis method returns the resulting data which we can then use for machine learning. The original data will still be accessible in the data attribute of the Feature Selector.\n\n**Be careful** of the methods used for removing features! It's a good idea to inspect the features that will be removed before using the remove function."},{"metadata":{"_uuid":"80a05b94cfc8197331fa9deae006b3b6612d053c","trusted":true},"cell_type":"code","source":"train_no_missing = fs.remove(methods = ['missing'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3618f1ce1acaf62a3d94f3adef16d4985b070d64","trusted":true},"cell_type":"code","source":"train_no_missing_zero = fs.remove(methods = ['missing', 'zero_importance'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"19ce6d1376a393428f4d97e753c45cc72fa3a1ac"},"cell_type":"markdown","source":"To remove the features from all of the methods, pass in method='all'. Before we do this, we can check how many features will be removed using check_removal. This returns a list of all the features that have been idenfitied for removal."},{"metadata":{"_uuid":"f629f55c75797694b6be03eee37e7fe5c56ba29c","trusted":true},"cell_type":"code","source":"all_to_remove = fs.check_removal()\nall_to_remove[10:25]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cdf4396867aaf83f75b57386cfdda30b765a7adc"},"cell_type":"markdown","source":"Now we can remove all of the features idenfitied."},{"metadata":{"_uuid":"5a97bf942e816f1ecc933f7709c2067adbe05ff2","trusted":true},"cell_type":"code","source":"train_removed = fs.remove(methods = 'all')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"df170ad96b32a9c3dce2935c6c6bb712a1ff5bc7"},"cell_type":"markdown","source":"## **Step.10 Handling One-Hot Features**\n\nIf we look at the dataframe that is returned, we may notice several new columns that were not in the original data. These are created when the data is one-hot encoded for machine learning. To remove all the one-hot features, we can pass in keep_one_hot = False to the remove method."},{"metadata":{"_uuid":"8cebbe9216b64e51790f88c1eee5f0a8666a0950","trusted":true},"cell_type":"code","source":"train_removed_all = fs.remove(methods = 'all', keep_one_hot=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"26a33b25fda33b0a01ed3c6c3994a0e11b365218","trusted":true},"cell_type":"code","source":"print('Original Number of Features :', train.shape[1])\nprint('Final Number of Features: ', train_removed_all.shape[1])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d872497ee4c731ace54d57a13fd74d4d98fbb9f7","trusted":true},"cell_type":"code","source":"train_removed_all.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e1c4f28321d2abbdb4718660ad43ce5c7702bee8","trusted":true},"cell_type":"code","source":"feature = train_removed_all.columns","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"033e40ffae211d1954122db03c49791ed49acad7","trusted":true},"cell_type":"code","source":"test_df = test[feature]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3b599e466ce75dd5958108a057a5049189048d5b","trusted":true},"cell_type":"code","source":"train_df = train_removed_all","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"28d9605a45433b0a5eff75c79beeee2b72b9d26a"},"cell_type":"markdown","source":"  #### Final shape"},{"metadata":{"_uuid":"ea393352b39b043aa58d4646d4f04d73fe9991fc","trusted":true},"cell_type":"code","source":"train_df.shape,labels.shape,test_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a72ffc4847779a56bf2a41cd68dc35c47a61de5"},"cell_type":"code","source":"col_tr = train_df.columns","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fe1c596187fcae8f1d19c1772930e19f99510623"},"cell_type":"markdown","source":"## **Step.11 Model Training**\n\n* Model is Binary Classification so used Logistic Regression which is best for all time."},{"metadata":{"trusted":true,"_uuid":"e86b02d2ab85f6863b2a386793b9386b2ee8ba04"},"cell_type":"code","source":"std = StandardScaler()\ntrain_df = std.fit_transform(train_df)\ntrain_df = pd.DataFrame(train_df,columns=col_tr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45254302b87badd32898d61d35a0c92f50a7b2de"},"cell_type":"code","source":"#Scaling Numerical columns\nstd = StandardScaler()\ntest_df = std.fit_transform(test_df)\ntest_df = pd.DataFrame(test_df,columns=col_tr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d50690344d079a61bced4bbef9779f4b63dec8b1"},"cell_type":"code","source":"train_df.shape,labels.shape,test_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7408c9ef0420dae94eb04bfb2a2df4bdaa162861"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(train_df, labels, test_size=0.20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e86babe6f537634d211e6c2d031f6c64512409cb"},"cell_type":"code","source":"kfold = model_selection.KFold(n_splits=10, random_state=42)\nbest_parameters = {'C': 0.1, 'class_weight': 'balanced', 'penalty': 'l1'}\n\nestimators = []\nmodel1 = LogisticRegression(solver=\"liblinear\"); est1 = RFE(model1, 25, step=1);estimators.append(('logistic', est1))\nmodel2 = DecisionTreeClassifier(); est2 = RFE(model1, 25, step=1);estimators.append(('cart', est2))\nmodel3 = SVC(); estimators.append(('svm', model3))\nensemble = VotingClassifier(estimators)\n# selector = RFE(ensemble, 25, step=1)\nresults = model_selection.cross_val_score(ensemble, X_train, y_train, cv=kfold)\nprint(); \nprint(\"Result Mean:{}\".format(results.mean()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"142c2de427e935f24385814d83ea20fbc9a5298e"},"cell_type":"code","source":"ensemble.fit(X_train, y_train)\nprint(\"Score:{0}\".format(ensemble.score(X_train, y_train)))\nprediction = ensemble.predict(test_df)\nsubmission = pd.read_csv(\"../input/sample_submission.csv\")\nsubmission['target'] = prediction\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"acdc62271036adc6b949c4bd476e23f97d9432e5"},"cell_type":"markdown","source":"## **Step.12 Model Evaluation Framework to check Importance**\n\n#### **Permutation importance**\n\n*There is also another way of using eli5 - we could have a look at permutation importance. It works in the following way:*\n* We fit a model;\n* We randomly shuffle one column of validation data and calculate the score;\n* If the score dropped significantly, it means that the feature is important;"},{"metadata":{"trusted":true,"_uuid":"6175fdf7a299e3a06ecf49937335bd6b1af5e6df"},"cell_type":"code","source":"perm = PermutationImportance(ensemble, random_state=42).fit(train_df, labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d4187e2859e665816be31e3a9df1cd26fb7707d"},"cell_type":"code","source":"submission = pd.read_csv(\"../input/sample_submission.csv\")\nsubmission['target'] = perm.predict(test_df)\nsubmission.to_csv('submission_perm.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a58af08b524db83ff5153b74f4510e84fce2eb25"},"cell_type":"code","source":"submission.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e60bde090888dd660e09deac0b23d87c2069ad0e"},"cell_type":"markdown","source":"## **Step.13 Model Training and Evaluation Framework to check Importance**"},{"metadata":{"trusted":true,"_uuid":"13947f6e0a9545b7677edc36700e524fcb1ad648"},"cell_type":"code","source":"model1 = LogisticRegression(class_weight='balanced', penalty='l1', C=0.1, solver='liblinear', tol=0.00001,dual=False)\nest1 = RFE(model1, 25, step=1)\nest1.fit(train_df,labels)\nprint(\"Score:{0}\".format(est1.score(train_df,labels)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04186dc51e5e19e53ce0b59a11a1b0ae523cbd95"},"cell_type":"code","source":"from mlxtend.feature_selection import SequentialFeatureSelector as SFS\nfrom mlxtend.plotting import plot_sequential_feature_selection as plot_sfs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0d091f72cdc5d249ae0cc93ee0a144eecaa602e"},"cell_type":"code","source":"sfs1 = SFS(est1, k_features=(10, 15), forward=True, floating=False,verbose=1,scoring='roc_auc',cv=5,n_jobs=-1)\nsfs1 = sfs1.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd5ca529433b31308c49f056458743635b4aa480"},"cell_type":"code","source":"plt.figure(figsize=(20,8))\nfig1 = plot_sfs(sfs1.get_metric_dict(), color = \"red\",kind='std_dev', marker=\"p\")\nplt.ylim([0.8, 1])\nplt.title('Sequential Forward Selection (w. StdDev)')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6106b796f2523a5ff8cba69652dd0fc4cdec77b"},"cell_type":"code","source":"sfseatures = list(sfs1.k_feature_names_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0efbc1d96629a94ed89c602f6bf3fb2b6be3b021"},"cell_type":"code","source":"train_1 = train_df[sfseatures]\ntest_1 = test_df[sfseatures]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf19bf82a0eeaf891d013fe33d814e37aa0364cb"},"cell_type":"code","source":"model1 = LogisticRegression(class_weight='balanced', penalty='l1', C=0.1, solver='liblinear', tol=0.00001,dual=False)\nest1 = RFE(model1, 25, step=1)\nest1.fit(train_1,labels)\nprint(\"Score:{0}\".format(est1.score(train_1,labels)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"356c5856c3522c6ffe2e9e9a96bc17441599eb8a"},"cell_type":"code","source":"submission = pd.read_csv(\"../input/sample_submission.csv\")\nsubmission['target'] = est1.predict(test_1)\nsubmission.to_csv('submission_sfs.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}