{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <b>1 <span style='color:#3f4d63'>|</span> Introduction</b>\n\n<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>1.1 | Table of contents</b></p>\n</div>\n\n* **<span style = 'color:red'>Exploratory Data Analysis</span>**\n    * **Basic information**\n    * **Distributions**\n        * Categorical Values\n        * Integers Values\n        * Floating Values\n        * Normal Distribution Test for Floating Values\n        * Distribution of dependent variable\n    * **Correlations**\n* **<span style = 'color:red'>Feature Engineering</span>**\n    * **Handling Categorical Features**\n    * **Handling Missing Values**\n    * **Feature Scaling**\n    * **Outlier Detection**\n* **<span style = 'color:red'>Feature Selection</span>**\n    * **Variance Threshold**\n        * Theory\n        * Code\n    * **Information Gain Method**\n        * Theory\n        * Code\n    * **ExtraTree**\n        * Code\n* **<span style = 'color:red'>Model</span>**\n    * **Preparing Test set**\n    * **GroupK Fold**\n    * **Combining Results**\n* **<span style = 'color:red'>Future Work</span>**","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nfrom scipy.stats import shapiro\nfrom termcolor import colored\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.feature_selection import VarianceThreshold\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.feature_selection import mutual_info_classif\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.model_selection import GroupKFold, StratifiedGroupKFold\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import roc_auc_score\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T06:12:48.789564Z","iopub.execute_input":"2022-08-04T06:12:48.790044Z","iopub.status.idle":"2022-08-04T06:12:49.768264Z","shell.execute_reply.started":"2022-08-04T06:12:48.789950Z","shell.execute_reply":"2022-08-04T06:12:49.767049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainDF = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntrainDF.drop('id', axis=1, inplace=True)\ntestDF = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\ntestDF.drop('id', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:49.770481Z","iopub.execute_input":"2022-08-04T06:12:49.770869Z","iopub.status.idle":"2022-08-04T06:12:50.055584Z","shell.execute_reply.started":"2022-08-04T06:12:49.770826Z","shell.execute_reply":"2022-08-04T06:12:50.054675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainPseudoDF = trainDF.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:50.056884Z","iopub.execute_input":"2022-08-04T06:12:50.057355Z","iopub.status.idle":"2022-08-04T06:12:50.063640Z","shell.execute_reply.started":"2022-08-04T06:12:50.057325Z","shell.execute_reply":"2022-08-04T06:12:50.062655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def infoDF(data):\n    floatColCounter = 0\n    floatCols = []\n    intColCounter = 0\n    intCols = []\n    stringColCounter = 0\n    stringCols = []\n    print('No of rows-> {}, No of columns-> {}'.format(data.shape[0], data.shape[1]))\n    print('          ------------------------          ')\n    for column in data.columns:\n        if data[column].dtype == int:\n            intColCounter += 1\n            print('{} dtype -> integer, % of null values-> {}%, No of distinct values-> {}'.format(column, round((data[column].isnull().sum()/data.shape[0])*100, 2), data[column].nunique()))\n            intCols.append(column)\n            print('          ------------------------          ')\n        elif data[column].dtype == float:\n            floatColCounter += 1\n            print('{} dtype -> float, % of null values-> {}%, No of distinct values-> {}'.format(column, round((data[column].isnull().sum()/data.shape[0])*100, 2), data[column].nunique()))\n            floatCols.append(column)\n            print('          ------------------------          ')\n        else:\n            stringColCounter += 1\n            print('{} dtype -> string, % of null values-> {}%, No of distinct values-> {}'.format(column, round((data[column].isnull().sum()/data.shape[0])*100, 2), data[column].nunique()))\n            stringCols.append(column)\n            print('          ------------------------          ')\n            \n    print('No of integer column-> {}, No of floating column-> {}, No of string or object columns-> {}'.format(intColCounter, floatColCounter, stringColCounter))\n    print('          ------------------------          ')\n    print('% of Null/Missing Values in data-> {}%'.format(round((data.isnull().sum().sum()/(data.shape[0]*trainDF.shape[1]))*100, 2)))\n    print('          ------------------------          ')\n    return intCols, floatCols, stringCols","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:50.066217Z","iopub.execute_input":"2022-08-04T06:12:50.066556Z","iopub.status.idle":"2022-08-04T06:12:50.079226Z","shell.execute_reply.started":"2022-08-04T06:12:50.066528Z","shell.execute_reply":"2022-08-04T06:12:50.078329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b>2 <span style='color:#3f4d63'>|</span> Exploratory Data Analysis</b>\n\n<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>2.1 | Basic information</b></p>\n</div>\n\nTo start with, we're gonna take a brief view on the dataset given in order to get some basic information about it: \n\n* We're gonna show some sample rows and columns of the dataframe.\n* Examine which type of features we've given.\n* Find out whether there are missing values or not.\n* No of distinct values of given feature.","metadata":{}},{"cell_type":"code","source":"intColumns, floatColumns, stringColumns = infoDF(trainDF)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:50.080633Z","iopub.execute_input":"2022-08-04T06:12:50.081407Z","iopub.status.idle":"2022-08-04T06:12:50.135103Z","shell.execute_reply.started":"2022-08-04T06:12:50.081360Z","shell.execute_reply":"2022-08-04T06:12:50.134263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>2.2 | Distributions</b></p>\n</div>\n\nIn this first section, we're gonna focus on analysing every feature's distribution and its related statistical information. To start with, let's plot the distributions in order to determine whether features are distributed normally. \n> We're gonna distinguish between different categorical features and we can se the distribution of all three categorical feature of train set.","metadata":{}},{"cell_type":"code","source":"figure = plt.figure(figsize=(16, 3))\ncolors = ['red', 'blue', 'green']\nfor i in range(len(stringColumns)):\n    plt.subplot(1, 3, i+1)\n    sns.histplot(trainDF[stringColumns[i]], shrink=0.8, color=colors[i])\nfigure.tight_layout(h_pad=1.0, w_pad=0.5)\nplt.suptitle('Distribution of String Column Values')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:50.136557Z","iopub.execute_input":"2022-08-04T06:12:50.137189Z","iopub.status.idle":"2022-08-04T06:12:50.783508Z","shell.execute_reply.started":"2022-08-04T06:12:50.137151Z","shell.execute_reply":"2022-08-04T06:12:50.782399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Below is the code for the distribution of the integer feautre and and the graph should show un-even spread as we have seen in basic information that all the integers values conatin 4-5(max) distinct values.","metadata":{}},{"cell_type":"code","source":"fig=plt.figure(figsize=(16, 5))\nfor i, f in enumerate(intColumns):\n    plt.style.use('ggplot')\n    plt.subplot(2, 4, i+1)\n    sns.histplot(trainDF[f])\n    plt.title('Feature: {}'.format(f))\n    plt.xlabel('')\n    \nfig.suptitle('Integer Feature distributions',  size=20)\nfig.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:50.785355Z","iopub.execute_input":"2022-08-04T06:12:50.786441Z","iopub.status.idle":"2022-08-04T06:12:52.520371Z","shell.execute_reply.started":"2022-08-04T06:12:50.786396Z","shell.execute_reply":"2022-08-04T06:12:52.519247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> After integer we look at floating value feature which should show different distribution of values integer value feature as there way more distinct feature and its also clear from graph. Most of the feature seen to be normally distributed.","metadata":{}},{"cell_type":"code","source":"fig=plt.figure(figsize=(16, 8))\nfor i, f in enumerate(floatColumns):\n    plt.style.use('ggplot')\n    plt.subplot(4, 4, i+1)\n    sns.histplot(trainDF[f])\n    plt.title('Feature: {}'.format(f))\n    plt.xlabel('')\n    \nfig.suptitle('Float Feature distributions',  size=20)\nfig.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:52.521776Z","iopub.execute_input":"2022-08-04T06:12:52.522137Z","iopub.status.idle":"2022-08-04T06:12:57.715144Z","shell.execute_reply.started":"2022-08-04T06:12:52.522106Z","shell.execute_reply":"2022-08-04T06:12:57.714085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"📌 **Early insights:**\n\n* It seems that every `float` feature is distributed normally. By contrast, that's not what happens when talking about `ìnt` features. \n\nThus, let's determine it. We can proceed with different methods. Let's show some of them: \n\n**Shapiro-Wilk Test** (Code taken from [TPS Jul 22 Advanced. Author: Torch me](https://www.kaggle.com/code/kartushovdanil/tps-jul-22-advanced-2-sol))\n\nThis test is used to test whether a dataset is distributed normally or not. The null hypothesis is that a sample $$x_1\\hspace{0.1cm},\\hspace{0.1cm}\\cdots\\hspace{0.1cm},\\hspace{0.1cm}x_n$$ comes from a normally distributed population. It was published in 1965 by Samuel Shapiro and Martin Wilk. **It is considered one of the most powerful tests for normality testing.** The test stadistic will be: \n\n$$W = \\frac{(\\sum_{i=1}^{n}a_{i}x_i)^2}{\\sum_{i=1}^{n}(x_i - \\bar{x})^2}$$\n\nwhere\n\n* $x_i$ is the number occupying the i-th position in the sample (with the sample ordered from smallest to largest).\n* $\\bar{x}$ is the sample mean. \n* Variables $a_i$ are calculated this way: \n\n$$(a_1, ... , a_n) = \\frac{m^T V^{-1}}{(m^T V^{-1}V^{-1}m)^{1/2}} \\hspace{2cm}m = (m_1 , ... , m_n)$$\n\nwhere $m_1 , ... , m_n$ are the mean values of the ordered statistic, of independent and identically distributed random variables, sampled from normal distributions and $V$ denotes the covariance matrix of that order statistic. **The null hypothesis is rejected if W is too small. The value of W can range from 0 to 1.**","metadata":{}},{"cell_type":"code","source":"for col in intColumns+floatColumns:\n    stat, p_value = shapiro(trainDF[col])  \n    alpha = 0.05\n    if p_value > alpha: \n        result = colored('Accepted', 'green')  \n    else:\n        result = colored('Rejected','red')        \n    print('Feature: {}\\t Hypothesis: {}'.format(col, result))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:57.716323Z","iopub.execute_input":"2022-08-04T06:12:57.717142Z","iopub.status.idle":"2022-08-04T06:12:57.775121Z","shell.execute_reply.started":"2022-08-04T06:12:57.717104Z","shell.execute_reply":"2022-08-04T06:12:57.774069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Finally distribution of dependent vairable and from the first look itself, it seem to be an imbalance in dependen feature which wee need to tackle in model creation.","metadata":{}},{"cell_type":"code","source":"sns.histplot(trainDF['failure'])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:57.779033Z","iopub.execute_input":"2022-08-04T06:12:57.779387Z","iopub.status.idle":"2022-08-04T06:12:58.052292Z","shell.execute_reply.started":"2022-08-04T06:12:57.779355Z","shell.execute_reply":"2022-08-04T06:12:58.051083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's convenient not to use features that are correlated (hence redundant), when trying to make a proper ML application. Thus, in this section, our main aim will be to analyse the different relationships between each of the features. Thus, we'll be able to determine which features are linearly related.\n\n📌 Insights:\n\nIt seems that attribute features are lightly correlated with measurement features.\nWhen talking about float features, if we recap which features are normally distributed and these features seem to  be correlated.\nMost correlated features are attribute_01 and attribute_00.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\ncorr = trainPseudoDF.corr()\nmatrix = np.triu(corr)\nsns.heatmap(corr, mask = matrix, center = 0, cmap = 'vlag').set_title('Correlations')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:58.053807Z","iopub.execute_input":"2022-08-04T06:12:58.054974Z","iopub.status.idle":"2022-08-04T06:12:58.660945Z","shell.execute_reply.started":"2022-08-04T06:12:58.054927Z","shell.execute_reply":"2022-08-04T06:12:58.660074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b>3 <span style='color:#3f4d63'>|</span> Feature Engineering</b>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>3.1 | Handling Categorical Values</b></p>\n</div>\n\nFirsly we will have glimpse on the values of the categorical feature so that we can come up with a strategy of handling categorical values.","metadata":{}},{"cell_type":"code","source":"for i in stringColumns:\n    print('Unique Values for {} -> {}'.format(i, trainDF[i].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:58.662276Z","iopub.execute_input":"2022-08-04T06:12:58.662884Z","iopub.status.idle":"2022-08-04T06:12:58.676064Z","shell.execute_reply.started":"2022-08-04T06:12:58.662840Z","shell.execute_reply":"2022-08-04T06:12:58.674865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **Proudct Type:** Product type seem to have 5 distinct values in train and 4 distinct values in test thus it not feasable to use any encoder technique to apply on product type as it will have no information of product code importance while predicting for test set.\n* **Attribute 0:** we can simply fetch number from their values by removing material and use it as final feature and alos same for test set feature as thses values contain in both train and test set.\n* **Attribute 0**: we can simply fetch number from their values by removing material and use it as final feature and alos same for test set feature as thses values contain in both train and test set.","metadata":{}},{"cell_type":"code","source":"trainPseudoDF['attribute_1'] = trainPseudoDF['attribute_1'].str.split('_', 1).str[1].astype('int')\ntrainPseudoDF['attribute_0'] = trainPseudoDF['attribute_0'].str.split('_', 1).str[1].astype('int')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:58.677647Z","iopub.execute_input":"2022-08-04T06:12:58.678587Z","iopub.status.idle":"2022-08-04T06:12:59.003043Z","shell.execute_reply.started":"2022-08-04T06:12:58.678548Z","shell.execute_reply":"2022-08-04T06:12:59.001770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>3.2 | Handling Missing Values</b></p>\n</div>\n\n📌 **Early insights:**\n\n* All the null values in float value features. \n* All the floating value feature are normally distributed\n* **Null Values/ Missing Values**: Because of the above insights replacing the missing values with mean seem to be most easy, less computaional and efficient solution.","metadata":{}},{"cell_type":"code","source":"for col in floatColumns:\n    if trainPseudoDF[col].isnull().sum():\n        trainPseudoDF[col].fillna(trainPseudoDF[col].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:59.004913Z","iopub.execute_input":"2022-08-04T06:12:59.005343Z","iopub.status.idle":"2022-08-04T06:12:59.026944Z","shell.execute_reply.started":"2022-08-04T06:12:59.005306Z","shell.execute_reply":"2022-08-04T06:12:59.026023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputer = KNNImputer(n_neighbors = 5)\n# trainPseudoDF[floatColumns] = imputer.fit_transform(trainPseudoDF[floatColumns])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:59.028424Z","iopub.execute_input":"2022-08-04T06:12:59.029219Z","iopub.status.idle":"2022-08-04T06:12:59.034259Z","shell.execute_reply.started":"2022-08-04T06:12:59.029181Z","shell.execute_reply":"2022-08-04T06:12:59.033147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainPseudoDF[floatColumns].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:59.035678Z","iopub.execute_input":"2022-08-04T06:12:59.036357Z","iopub.status.idle":"2022-08-04T06:12:59.055992Z","shell.execute_reply.started":"2022-08-04T06:12:59.036321Z","shell.execute_reply":"2022-08-04T06:12:59.055058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>3.3 | Feature Scaling</b></p>\n</div>\n\n📌 **Early insights:**\n* As we know, there is a drastic change in range of integer, floating values features and in categorical feature thus it semm feasible to bring all feature value in a single range this scaling feature seems to be a logical steps","metadata":{}},{"cell_type":"code","source":"scalerModel = StandardScaler().fit(trainPseudoDF.drop(['failure', 'product_code'], axis=1))\nscaledPseudoDF = scalerModel.transform(trainPseudoDF.drop(['failure', 'product_code'], axis=1))\nscaledPseudoDF = pd.DataFrame(scaledPseudoDF, columns=trainPseudoDF.drop(['failure', 'product_code'], axis=1).columns)\nscaledPseudoDF","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:59.057902Z","iopub.execute_input":"2022-08-04T06:12:59.058526Z","iopub.status.idle":"2022-08-04T06:12:59.124958Z","shell.execute_reply.started":"2022-08-04T06:12:59.058491Z","shell.execute_reply":"2022-08-04T06:12:59.123870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaledPseudoDF['product_code'] = trainPseudoDF['product_code']\nscaledPseudoDF['failure'] = trainPseudoDF['failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:59.126440Z","iopub.execute_input":"2022-08-04T06:12:59.126860Z","iopub.status.idle":"2022-08-04T06:12:59.134292Z","shell.execute_reply.started":"2022-08-04T06:12:59.126765Z","shell.execute_reply":"2022-08-04T06:12:59.133214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Lets see the correlation after scaling","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\ncorr = scaledPseudoDF.corr()\nmatrix = np.triu(corr)\nsns.heatmap(corr, mask = matrix, center = 0, cmap = 'vlag').set_title('Correlations')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:59.136028Z","iopub.execute_input":"2022-08-04T06:12:59.136409Z","iopub.status.idle":"2022-08-04T06:12:59.862123Z","shell.execute_reply.started":"2022-08-04T06:12:59.136378Z","shell.execute_reply":"2022-08-04T06:12:59.861032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>3.4 | Handling outliers</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"tmpDF = pd.DataFrame(data = scaledPseudoDF[intColumns].drop('failure', axis=1))\nplt.figure(figsize=(16,4)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(tmpDF)).set_title('Boxplot of each feature',size=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:12:59.864290Z","iopub.execute_input":"2022-08-04T06:12:59.865112Z","iopub.status.idle":"2022-08-04T06:13:00.201127Z","shell.execute_reply.started":"2022-08-04T06:12:59.865075Z","shell.execute_reply":"2022-08-04T06:13:00.200322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Floating value feature seems to have nore outliers compaired to integer value feature let more to find out about outlieries in floating value features.","metadata":{}},{"cell_type":"code","source":"tmpDF = pd.DataFrame(data = scaledPseudoDF[floatColumns])\nplt.figure(figsize=(16,4)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(tmpDF)).set_title('Boxplot of each feature',size=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:00.202176Z","iopub.execute_input":"2022-08-04T06:13:00.203011Z","iopub.status.idle":"2022-08-04T06:13:00.864922Z","shell.execute_reply.started":"2022-08-04T06:13:00.202966Z","shell.execute_reply":"2022-08-04T06:13:00.863624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in floatColumns:\n    outlierLen = scaledPseudoDF[(scaledPseudoDF[col] <= -3) | (scaledPseudoDF[col] >= 3)].shape[0]\n    outliersPercentage = round((outlierLen/len(scaledPseudoDF[col]))*100,2)\n    print('% of outliers in col {} -> {}%'.format(col, outliersPercentage))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:00.866224Z","iopub.execute_input":"2022-08-04T06:13:00.866546Z","iopub.status.idle":"2022-08-04T06:13:00.890535Z","shell.execute_reply.started":"2022-08-04T06:13:00.866517Z","shell.execute_reply":"2022-08-04T06:13:00.889458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"📌 Insights:\n* There is somehwere around 1% of outliers in almost all the floating value features thus we try to cut this % to half so that our model can have some information about the outliers also when predicting value for test set.","metadata":{}},{"cell_type":"code","source":"for col in floatColumns:\n    indexes = scaledPseudoDF[(scaledPseudoDF[col] <= -3) | (scaledPseudoDF[col] >= 3)].index\n    scaledPseudoDF.drop(indexes, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:00.892280Z","iopub.execute_input":"2022-08-04T06:13:00.893385Z","iopub.status.idle":"2022-08-04T06:13:00.979259Z","shell.execute_reply.started":"2022-08-04T06:13:00.893338Z","shell.execute_reply":"2022-08-04T06:13:00.978147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmpDF = pd.DataFrame(data = scaledPseudoDF[floatColumns])\nplt.figure(figsize=(16,4)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(tmpDF)).set_title('Boxplot of each feature',size=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:00.980986Z","iopub.execute_input":"2022-08-04T06:13:00.981446Z","iopub.status.idle":"2022-08-04T06:13:01.629660Z","shell.execute_reply.started":"2022-08-04T06:13:00.981402Z","shell.execute_reply":"2022-08-04T06:13:01.628531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b>4 <span style='color:#3f4d63'>|</span> Feature Selection</b>\n\n<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>4.1 | Variance Threshold</b></p>\n</div>\n\n* Feature selector that removes all low-variance features.\n\n* This feature selection algorithm looks only at the features (X), not the desired outputs (y).","metadata":{}},{"cell_type":"code","source":"selector = VarianceThreshold(threshold=1)\nselector.fit_transform(scaledPseudoDF.drop(['failure', 'product_code'], axis=1))\nplt.figure(figsize=(15,10))\nsns.barplot(x=selector.variances_, y=scaledPseudoDF.drop(['failure', 'product_code'], axis=1).columns,orient='h' ).set_title('Feature selection with VarianceThreshold',size=15);\nplt.xlabel('Variance');\nplt.axvline(x=.8, color='r', linestyle='--', label='Threshold')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:01.631366Z","iopub.execute_input":"2022-08-04T06:13:01.631844Z","iopub.status.idle":"2022-08-04T06:13:02.078618Z","shell.execute_reply.started":"2022-08-04T06:13:01.631782Z","shell.execute_reply":"2022-08-04T06:13:02.077445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>4.2 | Information Gain</b></p>\n</div>\n\n* MI Estimate mutual information for a discrete target variable.\n\n* Mutual information (MI) between two random variables is a non-negative value, which measures the dependency between the variables. It is equal to zero if and only if two random variables are independent, and higher values mean higher dependency.\n\n* The function relies on nonparametric methods based on entropy estimation from k-nearest neighbors distances.\n\n<b>Inshort<b>\n\n* A quantity called mutual information measures the amount of information one can obtain from one random variable given another.\n\n* The mutual information between two random variables X and Y can be stated formally as follows:\n\n<b>I(X ; Y) = H(X) – H(X | Y)<b>\nWhere I(X ; Y) is the mutual information for X and Y, H(X) is the entropy for X and H(X | Y) is the conditional entropy for X given Y. The result has the units of bits.","metadata":{}},{"cell_type":"code","source":"mutual_info = mutual_info_classif(scaledPseudoDF.drop(['failure', 'product_code'], axis=1),scaledPseudoDF['failure'])\nmutual_info = pd.Series(mutual_info)\nmutual_info.index = scaledPseudoDF.drop(['failure', 'product_code'], axis=1).columns\nmutual_info.sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:02.080025Z","iopub.execute_input":"2022-08-04T06:13:02.080442Z","iopub.status.idle":"2022-08-04T06:13:04.730175Z","shell.execute_reply.started":"2022-08-04T06:13:02.080403Z","shell.execute_reply":"2022-08-04T06:13:04.729069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>4.3 | Extra Tree</b></p>\n</div>\n\n* This technique gives you a score for each feature of your data,the higher the score mor relevant it is","metadata":{}},{"cell_type":"code","source":"model = ExtraTreesClassifier()\nmodel.fit(scaledPseudoDF.drop(['failure', 'product_code'], axis=1), scaledPseudoDF['failure'])\nplt.figure(figsize=(12, 6))\nsns.barplot(x=model.feature_importances_, y=scaledPseudoDF.drop(['failure', 'product_code'], axis=1).columns,orient='h' ).set_title('Feature selection with Extra Tree Classifier',size=15);\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:04.731335Z","iopub.execute_input":"2022-08-04T06:13:04.731640Z","iopub.status.idle":"2022-08-04T06:13:09.178174Z","shell.execute_reply.started":"2022-08-04T06:13:04.731613Z","shell.execute_reply":"2022-08-04T06:13:09.177031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"📌 Insights:\n\n* Floating values feature seem to be more important compaired to some integer values feature to predict the dependable feature outcome.\n* loading feature seems to be the best feature with highest importance across all features.","metadata":{}},{"cell_type":"markdown","source":"# <b>5 <span style='color:#3f4d63'>|</span> Model</b>\n\n<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>5.1 | Prepairing Test set</b></p>\n</div>\n\n* Prepairing test and performing all the transformation we did with train set so we can ready our test set for prediction.","metadata":{}},{"cell_type":"code","source":"_,_,_ = infoDF(testDF)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.185717Z","iopub.execute_input":"2022-08-04T06:13:09.186128Z","iopub.status.idle":"2022-08-04T06:13:09.227549Z","shell.execute_reply.started":"2022-08-04T06:13:09.186094Z","shell.execute_reply":"2022-08-04T06:13:09.226364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testPseudoDF = testDF.copy()\n\nfor col in floatColumns:\n    if testPseudoDF[col].isnull().sum():\n        testPseudoDF[col].fillna(testPseudoDF[col].mean(), inplace=True)\n\n# testPseudoDF[floatColumns] = imputer.fit_transform(testPseudoDF[floatColumns])\n# testPseudoDF[floatColumns].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.228754Z","iopub.execute_input":"2022-08-04T06:13:09.229107Z","iopub.status.idle":"2022-08-04T06:13:09.249707Z","shell.execute_reply.started":"2022-08-04T06:13:09.229077Z","shell.execute_reply":"2022-08-04T06:13:09.248786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in stringColumns:\n    print('Unique Values for {} -> {}'.format(i, testDF[i].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.250864Z","iopub.execute_input":"2022-08-04T06:13:09.251774Z","iopub.status.idle":"2022-08-04T06:13:09.262440Z","shell.execute_reply.started":"2022-08-04T06:13:09.251737Z","shell.execute_reply":"2022-08-04T06:13:09.261482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testPseudoDF['attribute_1'] = testPseudoDF['attribute_1'].str.split('_', 1).str[1].astype('int')\ntestPseudoDF['attribute_0'] = testPseudoDF['attribute_0'].str.split('_', 1).str[1].astype('int')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.263853Z","iopub.execute_input":"2022-08-04T06:13:09.264602Z","iopub.status.idle":"2022-08-04T06:13:09.333107Z","shell.execute_reply.started":"2022-08-04T06:13:09.264569Z","shell.execute_reply":"2022-08-04T06:13:09.331943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scalerModel = StandardScaler().fit(testPseudoDF.drop('product_code', axis=1))\nscaledtestPseudoDF = scalerModel.transform(testPseudoDF.drop('product_code', axis=1))\nscaledtestPseudoDF = pd.DataFrame(scaledtestPseudoDF, columns=testPseudoDF.drop('product_code', axis=1).columns)\nscaledtestPseudoDF['product_code'] = testPseudoDF['product_code']\nscaledtestPseudoDF","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.334706Z","iopub.execute_input":"2022-08-04T06:13:09.335912Z","iopub.status.idle":"2022-08-04T06:13:09.400000Z","shell.execute_reply.started":"2022-08-04T06:13:09.335873Z","shell.execute_reply":"2022-08-04T06:13:09.398824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xtrain = scaledPseudoDF.drop('failure', axis=1).reset_index(drop=True)\nytrain = scaledPseudoDF['failure'].reset_index(drop=True)\nXtest = scaledtestPseudoDF","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.401604Z","iopub.execute_input":"2022-08-04T06:13:09.402793Z","iopub.status.idle":"2022-08-04T06:13:09.416189Z","shell.execute_reply.started":"2022-08-04T06:13:09.402744Z","shell.execute_reply":"2022-08-04T06:13:09.414873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xtrain = Xtrain.drop(['attribute_0', 'attribute_1', 'attribute_2', 'attribute_3'], axis=1)\nXtest = Xtest.drop(['attribute_0', 'attribute_1', 'attribute_2', 'attribute_3'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.417524Z","iopub.execute_input":"2022-08-04T06:13:09.418138Z","iopub.status.idle":"2022-08-04T06:13:09.429203Z","shell.execute_reply.started":"2022-08-04T06:13:09.418099Z","shell.execute_reply":"2022-08-04T06:13:09.428048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def getScore(model, yval, yvalPred):\n    valScore = roc_auc_score(yval, yvalPred)\n    print(\"Model -> {}, Validation Score -> {}\".format(model, valScore))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.430769Z","iopub.execute_input":"2022-08-04T06:13:09.431848Z","iopub.status.idle":"2022-08-04T06:13:09.438588Z","shell.execute_reply.started":"2022-08-04T06:13:09.431782Z","shell.execute_reply":"2022-08-04T06:13:09.437337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictProbs = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.440159Z","iopub.execute_input":"2022-08-04T06:13:09.442192Z","iopub.status.idle":"2022-08-04T06:13:09.450022Z","shell.execute_reply.started":"2022-08-04T06:13:09.442154Z","shell.execute_reply":"2022-08-04T06:13:09.448764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>5.2 | Group K Fold</b></p>\n</div>\n\n* We want to create a classifier which predicts correct probabilities for previously unseen products. To validate such a classifier, we have to simulate this situation by splitting the data so that the validation set contains other products than the training set. The correct method is a five-fold cross-validation where every fold uses four products for training and the fifth product for validation (GroupKFold).\n\nFold 0: Train on products A, B, C, D; validate on E<br>\nFold 1: Train on products A, B, C, E; validate on D<br>\nFold 2: Train on products A, B, D, E; validate on C<br>\nFold 3: Train on products A, C, D, E; validate on B<br>\nFold 4: Train on products B, C, D, E; validate on A<br>\n\n* If you don't split your data with the GroupKFold on products, you'll get a data leak and inflated cross-validation scores. \n<b> **Idea by @AmbrosM** <b>","metadata":{}},{"cell_type":"code","source":"#gkf = GroupKFold(n_splits=5)\nsgkf = StratifiedGroupKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(sgkf.split(Xtrain, ytrain, Xtrain.product_code)):\n    print(f\"===== fold{fold} =====\")\n    \n    X_train = Xtrain.iloc[idx_tr][Xtest.columns]\n    X_valid = Xtrain.iloc[idx_va][Xtest.columns]\n    X_test = Xtest.copy()\n    y_train = ytrain.iloc[idx_tr]\n    y_valid = ytrain.iloc[idx_va]\n    \n    features = [f for f in X_train.columns if f != 'product_code']\n        # Logistic Regression\n    \n    lrModel = LogisticRegression().fit(X_train[features], y_train)\n    yValPred = lrModel.predict_proba(X_valid[features])[:,1]\n    getScore('Logistic Regression', y_valid, yValPred)  \n    predictProbs[f'LR_{fold}'] = lrModel.predict_proba(X_test[features])[:,1]\n    \n    del lrModel, yValPred\n    gc.collect()\n    \n        \n#     # SVM \n    \n#     svcModel = SVC(probability=True).fit(X_train[features], y_train)\n#     yValPred = svcModel.predict_proba(X_valid[features])[:,1]\n#     getScore('SVM', y_valid, yValPred)\n#     predictProbs[f'SVC_{fold}'] = svcModel.predict_proba(X_test[features])[:,1]\n    \n#     del svcModel, yValPred\n#     gc.collect()\n    \n#     # K Neighbours \n    \n#     knnModel = KNeighborsClassifier(n_neighbors=20).fit(X_train[features], y_train)\n#     yValPred = knnModel.predict_proba(X_valid[features])[:,1]\n#     getScore('KNN', y_valid, yValPred)\n#     predictProbs[f'KNN_{fold}'] = knnModel.predict_proba(X_test[features])[:,1]\n    \n#     del knnModel, yValPred\n#     gc.collect()\n    \n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:09.451443Z","iopub.execute_input":"2022-08-04T06:13:09.452194Z","iopub.status.idle":"2022-08-04T06:13:11.356982Z","shell.execute_reply.started":"2022-08-04T06:13:09.452160Z","shell.execute_reply":"2022-08-04T06:13:11.355862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>5.3 | Combining Results</b></p>\n</div>\n\n* Combining result of best fermormed model and submitting values","metadata":{}},{"cell_type":"code","source":"requiredPredCols = [col for col in predictProbs if 'LR' in col]\nfailure1 = predictProbs[requiredPredCols].sum(axis=1)/5","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:11.359660Z","iopub.execute_input":"2022-08-04T06:13:11.360549Z","iopub.status.idle":"2022-08-04T06:13:11.369469Z","shell.execute_reply.started":"2022-08-04T06:13:11.360501Z","shell.execute_reply":"2022-08-04T06:13:11.368134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\n# submission['failure'] = failure\n# submission.to_csv('submission.csv', index=False)\n# submission","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:13:11.371151Z","iopub.execute_input":"2022-08-04T06:13:11.371506Z","iopub.status.idle":"2022-08-04T06:13:11.377442Z","shell.execute_reply.started":"2022-08-04T06:13:11.371473Z","shell.execute_reply":"2022-08-04T06:13:11.376520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;\n            background-color:#3f4d6f;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>Ensembling Results</b></p>\n</div>\n","metadata":{}},{"cell_type":"code","source":"def inference(X, X_test, iterations):\n    pred_list = []\n    for i in range(iterations):\n        X_train = X.sample(int(0.8 * len(X)))\n        y_train = ytrain.loc[X_train.index]\n        \n        model = LogisticRegression(C = 0.0001, penalty = 'l2', random_state=i, tol = 1e-2, max_iter = 1000)\n        model.fit(X_train, y_train)\n        y_pred = model.predict_proba(X_test)[:,1]\n\n        pred_list.append(y_pred)    \n    pred_df = pd.DataFrame(pred_list).T\n    pred_df[\"mean\"] = pred_df.mean(axis=1)    \n    return pred_df['mean']\n\nfailure2 = inference(Xtrain.drop('product_code', axis=1), Xtest.drop('product_code', axis=1), iterations = 500)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:18:06.998000Z","iopub.execute_input":"2022-08-04T06:18:06.998481Z","iopub.status.idle":"2022-08-04T06:18:36.712821Z","shell.execute_reply.started":"2022-08-04T06:18:06.998443Z","shell.execute_reply":"2022-08-04T06:18:36.711782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\nsubmission['failure'] = failure1*0.5 + failure2*0.95\nsubmission.to_csv('submission.csv', index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-04T06:18:47.097485Z","iopub.execute_input":"2022-08-04T06:18:47.097938Z","iopub.status.idle":"2022-08-04T06:18:47.170562Z","shell.execute_reply.started":"2022-08-04T06:18:47.097902Z","shell.execute_reply":"2022-08-04T06:18:47.169434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}