{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport scipy.stats as stats\nimport missingno\n\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import FunctionTransformer, PowerTransformer, StandardScaler\nfrom sklearn.model_selection import cross_validate, StratifiedKFold\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T11:14:06.988841Z","iopub.execute_input":"2022-08-10T11:14:06.989267Z","iopub.status.idle":"2022-08-10T11:14:07.002143Z","shell.execute_reply.started":"2022-08-10T11:14:06.989235Z","shell.execute_reply":"2022-08-10T11:14:07.000612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Load data + overview","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv')\nsample_submission = pd.read_csv(\"/kaggle/input/tabular-playground-series-aug-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.004716Z","iopub.execute_input":"2022-08-10T11:14:07.005718Z","iopub.status.idle":"2022-08-10T11:14:07.248127Z","shell.execute_reply.started":"2022-08-10T11:14:07.005658Z","shell.execute_reply":"2022-08-10T11:14:07.246279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.250714Z","iopub.execute_input":"2022-08-10T11:14:07.251638Z","iopub.status.idle":"2022-08-10T11:14:07.286105Z","shell.execute_reply.started":"2022-08-10T11:14:07.251580Z","shell.execute_reply":"2022-08-10T11:14:07.284444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.287964Z","iopub.execute_input":"2022-08-10T11:14:07.290738Z","iopub.status.idle":"2022-08-10T11:14:07.325843Z","shell.execute_reply.started":"2022-08-10T11:14:07.290684Z","shell.execute_reply":"2022-08-10T11:14:07.324623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The train dataset has 26 columns including the target feature (failure). Most of the variables are numerical, need to check the correlation matrix (for feature selection). I'll drop the ID column since it does not provide any necessary information.","metadata":{}},{"cell_type":"code","source":"train = train.drop('id', axis = 1)\ntest = test.drop('id', axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.329222Z","iopub.execute_input":"2022-08-10T11:14:07.329963Z","iopub.status.idle":"2022-08-10T11:14:07.340103Z","shell.execute_reply.started":"2022-08-10T11:14:07.329916Z","shell.execute_reply":"2022-08-10T11:14:07.338919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.341571Z","iopub.execute_input":"2022-08-10T11:14:07.341916Z","iopub.status.idle":"2022-08-10T11:14:07.365416Z","shell.execute_reply.started":"2022-08-10T11:14:07.341884Z","shell.execute_reply":"2022-08-10T11:14:07.363906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.366971Z","iopub.execute_input":"2022-08-10T11:14:07.369943Z","iopub.status.idle":"2022-08-10T11:14:07.392332Z","shell.execute_reply.started":"2022-08-10T11:14:07.369901Z","shell.execute_reply":"2022-08-10T11:14:07.390766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['product_code'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.393981Z","iopub.execute_input":"2022-08-10T11:14:07.394967Z","iopub.status.idle":"2022-08-10T11:14:07.612895Z","shell.execute_reply.started":"2022-08-10T11:14:07.394927Z","shell.execute_reply":"2022-08-10T11:14:07.611704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['product_code'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.614894Z","iopub.execute_input":"2022-08-10T11:14:07.615283Z","iopub.status.idle":"2022-08-10T11:14:07.830175Z","shell.execute_reply.started":"2022-08-10T11:14:07.615247Z","shell.execute_reply":"2022-08-10T11:14:07.828850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.concat([train, test], axis = 0)\ndata['attribute_0'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.831837Z","iopub.execute_input":"2022-08-10T11:14:07.832192Z","iopub.status.idle":"2022-08-10T11:14:07.853960Z","shell.execute_reply.started":"2022-08-10T11:14:07.832161Z","shell.execute_reply":"2022-08-10T11:14:07.852865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['attribute_1'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.857163Z","iopub.execute_input":"2022-08-10T11:14:07.857555Z","iopub.status.idle":"2022-08-10T11:14:07.869756Z","shell.execute_reply.started":"2022-08-10T11:14:07.857523Z","shell.execute_reply":"2022-08-10T11:14:07.868518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['attribute_0', 'attribute_1']\nfor col in cols:\n    train[col] = [train[col][i].split('_')[1] for i in range(len(train))]\n    test[col] = [test[col][i].split('_')[1] for i in range(len(test))]\ntrain.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:07.871277Z","iopub.execute_input":"2022-08-10T11:14:07.872185Z","iopub.status.idle":"2022-08-10T11:14:08.568488Z","shell.execute_reply.started":"2022-08-10T11:14:07.872134Z","shell.execute_reply":"2022-08-10T11:14:08.567136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is imbalanced (failure is less common than proper functionality, which is a normal thing).","metadata":{}},{"cell_type":"code","source":"train['failure'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:08.569890Z","iopub.execute_input":"2022-08-10T11:14:08.570210Z","iopub.status.idle":"2022-08-10T11:14:08.804777Z","shell.execute_reply.started":"2022-08-10T11:14:08.570181Z","shell.execute_reply":"2022-08-10T11:14:08.803360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Fill NA values","metadata":{}},{"cell_type":"markdown","source":"There are missing values in the dataset, need to impute. I'll use KNN imputer.","metadata":{}},{"cell_type":"code","source":"print(train['measurement_11'].isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:08.806222Z","iopub.execute_input":"2022-08-10T11:14:08.806608Z","iopub.status.idle":"2022-08-10T11:14:08.814175Z","shell.execute_reply.started":"2022-08-10T11:14:08.806573Z","shell.execute_reply":"2022-08-10T11:14:08.812867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(round(100*train.isna().sum()/len(train), 2))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:08.815894Z","iopub.execute_input":"2022-08-10T11:14:08.816396Z","iopub.status.idle":"2022-08-10T11:14:08.836046Z","shell.execute_reply.started":"2022-08-10T11:14:08.816327Z","shell.execute_reply":"2022-08-10T11:14:08.834899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(round(100*test.isna().sum()/len(test), 2))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:08.838511Z","iopub.execute_input":"2022-08-10T11:14:08.839487Z","iopub.status.idle":"2022-08-10T11:14:08.854690Z","shell.execute_reply.started":"2022-08-10T11:14:08.839437Z","shell.execute_reply":"2022-08-10T11:14:08.853220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Missing values in the train dataset: {round(100*train.isna().sum().sum()/(train.shape[0]*train.shape[1]), 3)}%')\nprint(f'Missing values in the train dataset: {round(100*test.isna().sum().sum()/(test.shape[0]*test.shape[1]), 3)}%')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:08.856514Z","iopub.execute_input":"2022-08-10T11:14:08.856876Z","iopub.status.idle":"2022-08-10T11:14:08.876955Z","shell.execute_reply.started":"2022-08-10T11:14:08.856844Z","shell.execute_reply":"2022-08-10T11:14:08.875536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missingno.matrix(df=train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:08.878774Z","iopub.execute_input":"2022-08-10T11:14:08.879127Z","iopub.status.idle":"2022-08-10T11:14:09.770824Z","shell.execute_reply.started":"2022-08-10T11:14:08.879095Z","shell.execute_reply":"2022-08-10T11:14:09.769745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_cols = [col for col in train.columns if train[col].isna().sum() != 0]\nprint(missing_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:09.771900Z","iopub.execute_input":"2022-08-10T11:14:09.772848Z","iopub.status.idle":"2022-08-10T11:14:09.791578Z","shell.execute_reply.started":"2022-08-10T11:14:09.772812Z","shell.execute_reply":"2022-08-10T11:14:09.790563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_cols_test = [col for col in test.columns if test[col].isna().sum() != 0]\nprint(missing_cols_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:09.792705Z","iopub.execute_input":"2022-08-10T11:14:09.793374Z","iopub.status.idle":"2022-08-10T11:14:09.813562Z","shell.execute_reply.started":"2022-08-10T11:14:09.793314Z","shell.execute_reply":"2022-08-10T11:14:09.812299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfailure_list = train['failure']\n\ntrain_ = train.drop(['product_code', 'failure'], axis = 1)\ntest_ = test.drop('product_code', axis = 1)\nimputer = KNNImputer(\n    n_neighbors = 5,\n    weights = 'distance',\n    metric = 'nan_euclidean',\n)\n\nimputer.fit(train_)\ntrain_t = imputer.transform(train_)\ntest_t = imputer.transform(test_)\ntrain_t = pd.DataFrame(train_t, columns = train_.columns)\ntest_t = pd.DataFrame(test_t, columns = test_.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:14:09.814741Z","iopub.execute_input":"2022-08-10T11:14:09.815407Z","iopub.status.idle":"2022-08-10T11:15:11.127875Z","shell.execute_reply.started":"2022-08-10T11:14:09.815322Z","shell.execute_reply":"2022-08-10T11:15:11.126647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_t['failure'] = failure_list\ntrain_t.head(15)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:11.129709Z","iopub.execute_input":"2022-08-10T11:15:11.130474Z","iopub.status.idle":"2022-08-10T11:15:11.173335Z","shell.execute_reply.started":"2022-08-10T11:15:11.130425Z","shell.execute_reply":"2022-08-10T11:15:11.172004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_t.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:11.175456Z","iopub.execute_input":"2022-08-10T11:15:11.175924Z","iopub.status.idle":"2022-08-10T11:15:11.215422Z","shell.execute_reply.started":"2022-08-10T11:15:11.175877Z","shell.execute_reply":"2022-08-10T11:15:11.214355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Exploratory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"Checking the statistic parameters of each variable in the dataset using `pd.describe()`.","metadata":{}},{"cell_type":"code","source":"train_t.describe().T.style.background_gradient(subset=['mean'], cmap='YlOrRd')\\\n                          .background_gradient(subset=['std'], cmap='PuBu')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:11.216721Z","iopub.execute_input":"2022-08-10T11:15:11.217773Z","iopub.status.idle":"2022-08-10T11:15:11.326699Z","shell.execute_reply.started":"2022-08-10T11:15:11.217730Z","shell.execute_reply":"2022-08-10T11:15:11.325587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next I will plot for each variable its histogram and a Q-Q plot to test if it is normally distributed. If so, the values of the variable should fall in a 45 degree line when ploted against the theoretical quantiles.","metadata":{}},{"cell_type":"code","source":"cols = train_t.drop('failure', axis = 1).columns\nnum_cols = len(cols)\nfigure = plt.figure(figsize=(30, 120))\nfor i in range(num_cols):\n    feature = cols[i]\n    plt.subplot(num_cols, 2, 2*i+1)\n    plt.hist(train_t[feature], bins=100)\n    plt.title(f'{feature}')\n    plt.subplot(num_cols, 2, 2*i+2)\n    stats.probplot(train_t[feature], dist='norm', plot = plt)\n# figure.tight_layout(h_pad=1.0, w_pad=1.0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:11.328228Z","iopub.execute_input":"2022-08-10T11:15:11.328852Z","iopub.status.idle":"2022-08-10T11:15:23.587437Z","shell.execute_reply.started":"2022-08-10T11:15:11.328814Z","shell.execute_reply":"2022-08-10T11:15:23.585802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the variables with high cardinality seem to be normally distributed. However, some variables such as *measurement_13* feature outliers, and other variables such as *loading* have skewed distributions. Need to apply transformations to modify their distributions. Need to apply outliers capping methods.","metadata":{}},{"cell_type":"markdown","source":"[Yeo-Johnson transformation](https://www.stat.umn.edu/arc/yjpower.pdf) is an adaptation of Box-Cox transformation that can also be used on neagtive values.","metadata":{}},{"cell_type":"code","source":"transformer = PowerTransformer(method='yeo-johnson', standardize = False)\ncols_to_transform = ['loading', 'measurement_0', 'measurement_1', 'measurement_2']\ntransformer.fit(train_t[cols_to_transform])\nnormal_data = transformer.transform(train_t[cols_to_transform])\nnormal_data = pd.DataFrame(normal_data, columns = cols_to_transform)\nnum_cols = len(cols_to_transform)\nfigure = plt.figure(figsize=(20, 20))\nfor i in range(num_cols):\n    feature = cols_to_transform[i]\n    plt.subplot(num_cols, 2, 2*i+1)\n    plt.hist(normal_data[feature], bins=100)\n    plt.title(f'{feature}')\n    plt.subplot(num_cols, 2, 2*i+2)\n    stats.probplot(normal_data[feature], dist='norm', plot = plt)\n# figure.tight_layout(h_pad=1.0, w_pad=1.0)\nplt.show()\n\ntrain_t[cols_to_transform] = normal_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:23.589209Z","iopub.execute_input":"2022-08-10T11:15:23.589717Z","iopub.status.idle":"2022-08-10T11:15:25.675714Z","shell.execute_reply.started":"2022-08-10T11:15:23.589675Z","shell.execute_reply":"2022-08-10T11:15:25.674459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the test samples come from the same distribution as the train samples, I will apply the same transformation on the test dataset. ","metadata":{}},{"cell_type":"code","source":"normal_data = transformer.transform(test_t[cols_to_transform])\nnormal_data = pd.DataFrame(normal_data, columns = cols_to_transform)\nnum_cols = len(cols_to_transform)\nfigure = plt.figure(figsize=(20, 20))\nfor i in range(num_cols):\n    feature = cols_to_transform[i]\n    plt.subplot(num_cols, 2, 2*i+1)\n    plt.hist(normal_data[feature], bins=100)\n    plt.title(f'{feature}')\n    plt.subplot(num_cols, 2, 2*i+2)\n    stats.probplot(normal_data[feature], dist='norm', plot = plt)\n# figure.tight_layout(h_pad=1.0, w_pad=1.0)\nplt.show()\n\ntest_t[cols_to_transform] = normal_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:25.677161Z","iopub.execute_input":"2022-08-10T11:15:25.677532Z","iopub.status.idle":"2022-08-10T11:15:27.618585Z","shell.execute_reply.started":"2022-08-10T11:15:25.677497Z","shell.execute_reply":"2022-08-10T11:15:27.617306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Time to check the correlation matrix. There is a ceratin degree of correlation between *attribute_1* and *attribute_0* as well as between *attribute_2* and *attribute_3*. I'll combine each pair of variables into a single variable (topic discused in [Des' notebook](https://www.kaggle.com/code/desalegngeb/tps08-logisticregression-and-some-fe)). ","metadata":{}},{"cell_type":"code","source":"corrMatrix = train_t.corr()\nplt.figure(figsize=(25,12))\nsns.heatmap(corrMatrix, cmap='BrBG', annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:27.620004Z","iopub.execute_input":"2022-08-10T11:15:27.620367Z","iopub.status.idle":"2022-08-10T11:15:31.081743Z","shell.execute_reply.started":"2022-08-10T11:15:27.620315Z","shell.execute_reply":"2022-08-10T11:15:31.080539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_t['attribute_0*1'] = train_t['attribute_0'] * train_t['attribute_1']\ntrain_t['attribute_2*3'] = train_t['attribute_2'] * train_t['attribute_3']\n\ntest_t['attribute_0*1'] = test_t['attribute_0'] * test_t['attribute_1']\ntest_t['attribute_2*3'] = test_t['attribute_2'] * test_t['attribute_3']\n\ntrain_t = train_t.drop([f'attribute_{i}' for i in range(4)], axis = 1)\ntest_t = test_t.drop([f'attribute_{i}' for i in range(4)], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:31.089334Z","iopub.execute_input":"2022-08-10T11:15:31.089886Z","iopub.status.idle":"2022-08-10T11:15:31.111707Z","shell.execute_reply.started":"2022-08-10T11:15:31.089829Z","shell.execute_reply":"2022-08-10T11:15:31.110451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meas_gr1_cols = [f\"measurement_{i:d}\" for i in list(range(3, 5)) + list(range(9, 17))]\ntrain_t['meas_gr1_avg'] = np.mean(train_t[meas_gr1_cols], axis=1)\ntrain_t['meas_gr1_std'] = np.std(train_t[meas_gr1_cols], axis=1)\n\ntest_t['meas_gr1_avg'] = np.mean(test_t[meas_gr1_cols], axis=1)\ntest_t['meas_gr1_std'] = np.std(test_t[meas_gr1_cols], axis=1) \n\nmeas_gr2_cols = [f\"measurement_{i:d}\" for i in list(range(5, 9))]\ntrain_t['meas_gr2_avg'] = np.mean(train_t[meas_gr2_cols], axis=1)\ntest_t['meas_gr2_avg'] = np.mean(test_t[meas_gr2_cols], axis=1)\n\ntrain_t['meas17/meas_gr2_avg'] = train_t['measurement_17'] / train_t['meas_gr2_avg']\ntest_t['meas17/meas_gr2_avg'] = test_t['measurement_17'] / test_t['meas_gr2_avg']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:31.113358Z","iopub.execute_input":"2022-08-10T11:15:31.113707Z","iopub.status.idle":"2022-08-10T11:15:31.161174Z","shell.execute_reply.started":"2022-08-10T11:15:31.113677Z","shell.execute_reply":"2022-08-10T11:15:31.160076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_t = train_t.drop([f'measurement_{i}' for i in list(range(3,18))], axis = 1)\ntest_t = test_t.drop([f'measurement_{i}' for i in list(range(3,18))], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:31.162908Z","iopub.execute_input":"2022-08-10T11:15:31.163286Z","iopub.status.idle":"2022-08-10T11:15:31.178148Z","shell.execute_reply.started":"2022-08-10T11:15:31.163252Z","shell.execute_reply":"2022-08-10T11:15:31.177016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_t.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:31.180132Z","iopub.execute_input":"2022-08-10T11:15:31.180670Z","iopub.status.idle":"2022-08-10T11:15:31.203405Z","shell.execute_reply.started":"2022-08-10T11:15:31.180614Z","shell.execute_reply":"2022-08-10T11:15:31.202522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_t.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:31.204539Z","iopub.execute_input":"2022-08-10T11:15:31.205642Z","iopub.status.idle":"2022-08-10T11:15:31.221816Z","shell.execute_reply.started":"2022-08-10T11:15:31.205599Z","shell.execute_reply":"2022-08-10T11:15:31.220690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Handling the outliers","metadata":{}},{"cell_type":"markdown","source":"Since most of the variables are now normally distributed, I will cap the distributions at a maximum/minimum value. In a normal distribution, ~99% of the values are in $[\\mu - 3*\\sigma, \\mu + 3*\\sigma]$, where $\\mu$ is the mean of the distribution and $\\sigma$ is the standard deviation.","metadata":{}},{"cell_type":"code","source":"dummy_df = pd.DataFrame(data = train_t, columns = train_t.columns[:15])\nplt.figure(figsize=(18,9)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(dummy_df)).set_title('Boxplot of each variable',size=12.5)\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:31.224121Z","iopub.execute_input":"2022-08-10T11:15:31.224643Z","iopub.status.idle":"2022-08-10T11:15:31.719092Z","shell.execute_reply.started":"2022-08-10T11:15:31.224595Z","shell.execute_reply":"2022-08-10T11:15:31.717688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_boundaries(df, feature):\n    upper = df[feature].mean() + 3 * df[feature].std()\n    lower = df[feature].mean() - 3 * df[feature].std()\n    \n    return upper, lower\n\nfor col in train_t.columns:\n    upper_l, lower_l = find_boundaries(train_t, col)\n    train_t[col] = np.where(train_t[col] > upper_l, upper_l, np.where(train_t[col] < lower_l, lower_l, train_t[col]))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:31.721018Z","iopub.execute_input":"2022-08-10T11:15:31.722175Z","iopub.status.idle":"2022-08-10T11:15:31.753378Z","shell.execute_reply.started":"2022-08-10T11:15:31.722132Z","shell.execute_reply":"2022-08-10T11:15:31.752108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_df = pd.DataFrame(data = train_t, columns = train_t.columns[:15])\nplt.figure(figsize=(18,9)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(dummy_df)).set_title('Boxplot of each variable',size=12.5)\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:31.755229Z","iopub.execute_input":"2022-08-10T11:15:31.755626Z","iopub.status.idle":"2022-08-10T11:15:32.279210Z","shell.execute_reply.started":"2022-08-10T11:15:31.755588Z","shell.execute_reply":"2022-08-10T11:15:32.277830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in test_t.columns:\n    upper_l, lower_l = find_boundaries(test_t, col)\n    test_t[col] = np.where(test_t[col] > upper_l, upper_l, np.where(test_t[col] < lower_l, lower_l, test_t[col]))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:32.280977Z","iopub.execute_input":"2022-08-10T11:15:32.282069Z","iopub.status.idle":"2022-08-10T11:15:32.307293Z","shell.execute_reply.started":"2022-08-10T11:15:32.282021Z","shell.execute_reply":"2022-08-10T11:15:32.306060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Feature Scaling","metadata":{}},{"cell_type":"markdown","source":"This is the final step in the dataset preprocessing. Scaling is one of the most important steps in processing the data, since it treats problems such as outliers or large magnitude (values of *measurement_17* are much bigger than the other variables' values). One of the most popular scaler used in data science is the standard scaler. ","metadata":{}},{"cell_type":"code","source":"failure_list = train_t['failure']\ntrain_t = train_t.drop('failure', axis = 1)\n\nscaler = StandardScaler()\n\nscaler.fit(train_t)\n\ntrain_scaled = scaler.transform(train_t)\ntest_scaled = scaler.transform(test_t)\n\ntrain_t = pd.DataFrame(train_scaled, columns = train_t.columns)\ntest_t = pd.DataFrame(test_scaled, columns = test_t.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:32.309233Z","iopub.execute_input":"2022-08-10T11:15:32.310037Z","iopub.status.idle":"2022-08-10T11:15:32.328888Z","shell.execute_reply.started":"2022-08-10T11:15:32.310000Z","shell.execute_reply":"2022-08-10T11:15:32.327782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_df = pd.DataFrame(data = train_t, columns = train_t.columns[:15])\nplt.figure(figsize=(18,9)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(dummy_df)).set_title('Boxplot of each variable',size=12.5)\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:32.330541Z","iopub.execute_input":"2022-08-10T11:15:32.331184Z","iopub.status.idle":"2022-08-10T11:15:32.786195Z","shell.execute_reply.started":"2022-08-10T11:15:32.331149Z","shell.execute_reply":"2022-08-10T11:15:32.784799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Model - Logistic Regression","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:33:01.938106Z","iopub.execute_input":"2022-08-09T12:33:01.939387Z","iopub.status.idle":"2022-08-09T12:33:01.945395Z","shell.execute_reply.started":"2022-08-09T12:33:01.939329Z","shell.execute_reply":"2022-08-09T12:33:01.943916Z"}}},{"cell_type":"markdown","source":"Logistic regression works best with normalized data. It is not a robust model to outliers.","metadata":{}},{"cell_type":"code","source":"%%time\nX, y = train_t, failure_list\n\nseed = 0\nfold = 5\nskf = StratifiedKFold(n_splits=fold, shuffle=True, random_state=seed)\n\nmodel = LogisticRegression(max_iter = 1000, C=0.001, penalty='l2', solver='newton-cg')\nscores = pd.DataFrame(cross_validate(model, X, y, scoring=['roc_auc'], cv=skf, return_train_score=True)).T\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:32.788257Z","iopub.execute_input":"2022-08-10T11:15:32.789198Z","iopub.status.idle":"2022-08-10T11:15:33.600768Z","shell.execute_reply.started":"2022-08-10T11:15:32.789151Z","shell.execute_reply":"2022-08-10T11:15:33.599318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X, y)\ny_pred = model.predict_proba(test_t)[:, 1]\nprint(y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:33.602832Z","iopub.execute_input":"2022-08-10T11:15:33.603715Z","iopub.status.idle":"2022-08-10T11:15:33.776973Z","shell.execute_reply.started":"2022-08-10T11:15:33.603666Z","shell.execute_reply":"2022-08-10T11:15:33.775736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame({'id': sample_submission['id'], 'failure': y_pred})\nresults.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T11:15:33.779213Z","iopub.execute_input":"2022-08-10T11:15:33.780144Z","iopub.status.idle":"2022-08-10T11:15:33.905818Z","shell.execute_reply.started":"2022-08-10T11:15:33.780086Z","shell.execute_reply":"2022-08-10T11:15:33.904593Z"},"trusted":true},"execution_count":null,"outputs":[]}]}