{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-08T10:20:40.217504Z","iopub.execute_input":"2022-08-08T10:20:40.217976Z","iopub.status.idle":"2022-08-08T10:20:40.226278Z","shell.execute_reply.started":"2022-08-08T10:20:40.217935Z","shell.execute_reply":"2022-08-08T10:20:40.225321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import libraries\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nsns.set_theme(style = \"whitegrid\")\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.preprocessing import RobustScaler\nfrom itertools import chain\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn import linear_model, decomposition\nfrom scipy.stats import spearmanr\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:40.232151Z","iopub.execute_input":"2022-08-08T10:20:40.233312Z","iopub.status.idle":"2022-08-08T10:20:40.244365Z","shell.execute_reply.started":"2022-08-08T10:20:40.233261Z","shell.execute_reply":"2022-08-08T10:20:40.243306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\ndf_train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ndf_train['type'] = 'train'\ndf_test['type'] = 'test'\ndf_all = pd.concat([df_train, df_test], axis = 0, ignore_index = True)\ndf_all.set_index('id', inplace = True)\ndf_all.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:40.246695Z","iopub.execute_input":"2022-08-08T10:20:40.247481Z","iopub.status.idle":"2022-08-08T10:20:40.499955Z","shell.execute_reply.started":"2022-08-08T10:20:40.247433Z","shell.execute_reply":"2022-08-08T10:20:40.498596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_all.info())\n\n# make list of numeric and categorical variables\ndf_all.dtypes\nnumeric_vars = [i for i in df_all.columns if df_all.dtypes[i] != 'object']\ncategorical_vars = [i for i in df_all.columns if df_all.dtypes[i] == 'object']\ncategorical_vars.remove('type')\nnumeric_vars.remove('failure')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:40.502821Z","iopub.execute_input":"2022-08-08T10:20:40.503291Z","iopub.status.idle":"2022-08-08T10:20:40.542634Z","shell.execute_reply.started":"2022-08-08T10:20:40.503241Z","shell.execute_reply":"2022-08-08T10:20:40.541409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all[numeric_vars].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:40.543968Z","iopub.execute_input":"2022-08-08T10:20:40.544412Z","iopub.status.idle":"2022-08-08T10:20:40.683282Z","shell.execute_reply.started":"2022-08-08T10:20:40.544369Z","shell.execute_reply":"2022-08-08T10:20:40.681927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check all unique values of categorical variables\nfor i in categorical_vars:\n    print(i, ': ', df_all[i].unique()) ","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:40.686859Z","iopub.execute_input":"2022-08-08T10:20:40.687407Z","iopub.status.idle":"2022-08-08T10:20:40.706288Z","shell.execute_reply.started":"2022-08-08T10:20:40.687357Z","shell.execute_reply":"2022-08-08T10:20:40.704526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:40.707776Z","iopub.execute_input":"2022-08-08T10:20:40.708136Z","iopub.status.idle":"2022-08-08T10:20:40.739950Z","shell.execute_reply.started":"2022-08-08T10:20:40.708102Z","shell.execute_reply":"2022-08-08T10:20:40.738815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Exploratory data analysis","metadata":{}},{"cell_type":"code","source":"# distribution of numerical features \ndf = pd.melt(df_all, value_vars = [x for x in numeric_vars])\ng = sns.FacetGrid(df, col = \"variable\", col_wrap = 3, sharey = False, sharex = False)\ng = g.map(sns.histplot, \"value\", bins = 20)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:40.741038Z","iopub.execute_input":"2022-08-08T10:20:40.741799Z","iopub.status.idle":"2022-08-08T10:20:48.573117Z","shell.execute_reply.started":"2022-08-08T10:20:40.741761Z","shell.execute_reply":"2022-08-08T10:20:48.571790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all categorical variables vs. failure\nfor i in range(len(categorical_vars)):\n    crosstb = pd.crosstab(df_all.failure, df_all[categorical_vars[i]]) # Creating crosstab\n    barplot = crosstb.plot.bar(rot = 0) # Creating barplot","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:48.574858Z","iopub.execute_input":"2022-08-08T10:20:48.575333Z","iopub.status.idle":"2022-08-08T10:20:49.218830Z","shell.execute_reply.started":"2022-08-08T10:20:48.575289Z","shell.execute_reply":"2022-08-08T10:20:49.217557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all numeric variables vs. failure\ndf1 = df_all[[x for x in numeric_vars if x not in ['attribute_2', 'attribute_3', 'measurement_0', 'measurement_1',\n                                                   'measurement_2']]]\nfor i, col in enumerate(df1.columns):\n    plt.figure(i)\n    sns.histplot(x = col, data = df1, hue = df_all.failure, kde = True, bins = 30)\n    \n\ndf1 = df_all[['attribute_2', 'attribute_3', 'measurement_0', 'measurement_1',\n                                                   'measurement_2']]\nfor j, col in enumerate(df1.columns):\n    plt.figure(i + j)\n    sns.histplot(x = col, data = df1, hue = df_all.failure)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:20:49.220409Z","iopub.execute_input":"2022-08-08T10:20:49.220806Z","iopub.status.idle":"2022-08-08T10:21:00.244399Z","shell.execute_reply.started":"2022-08-08T10:20:49.220773Z","shell.execute_reply":"2022-08-08T10:21:00.243209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Multicolinearity\n\n# The two following tables show absolute values of Pearson and Spearman correlation \n# coefficients between all pairs of explanatory numeric variables sorted from highest \n# to lowets correlation.\n\n# Pearson\nPearson_corr = df_all[[x for x in numeric_vars]].corr().abs()\nPearson_corr = Pearson_corr.unstack()\nPearson_corr = Pearson_corr.sort_values(kind = \"quicksort\", ascending = False)\nPearson_corr = Pearson_corr[Pearson_corr != 1].drop_duplicates()\n\nfeature1 = []\nfor pair in list(Pearson_corr.index):\n    feature1.append(pair[0])\n    \nfeature2 = []\nfor pair in list(Pearson_corr.index):\n    feature2.append(pair[1])\n    \nPearson_corr = pd.DataFrame(Pearson_corr)\nPearson_corr.reset_index(drop = True, inplace = True)\nPearson_corr.rename(columns = {0: \"Pearson\"}, inplace = True)\nPearson_corr['Feature1'] = feature1\nPearson_corr['Feature2'] = feature2    \nPearson_corr = Pearson_corr[['Feature1', 'Feature2', 'Pearson']]\nprint(Pearson_corr[:10])\n\n# Spearman\nSpearman_corr = []\nfor i in [x for x in numeric_vars]:\n    for j in [x for x in numeric_vars]:\n        if i != j:\n            \n            coef, p = spearmanr(df_all[i], df_all[j])\n            #calculate Spearmann correlation coefficient \n            Spearman_corr.append([i, j, abs(coef)])\n            \nSpearman_corr = pd.DataFrame(Spearman_corr)\nSpearman_corr.rename(columns={0: \"Feature1\", 1: \"Feature2\", 2: \"Spearman\"}, inplace = True)\nSpearman_corr.sort_values('Spearman', ascending = False, inplace = True)\nSpearman_corr.reset_index(drop = True, inplace = True)\nSpearman_corr = Spearman_corr.iloc[::2, :]\nprint('\\n', Spearman_corr[:10])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:21:00.245926Z","iopub.execute_input":"2022-08-08T10:21:00.246331Z","iopub.status.idle":"2022-08-08T10:21:00.661573Z","shell.execute_reply.started":"2022-08-08T10:21:00.246293Z","shell.execute_reply":"2022-08-08T10:21:00.660333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:21:00.665880Z","iopub.execute_input":"2022-08-08T10:21:00.666336Z","iopub.status.idle":"2022-08-08T10:21:00.675474Z","shell.execute_reply.started":"2022-08-08T10:21:00.666296Z","shell.execute_reply":"2022-08-08T10:21:00.673675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count and percent of missing values\ncount = df_all.isna().sum()[df_all.isna().sum() > 0]\npct = df_all.isna().sum()[df_all.isna().sum() > 0] / df_all.shape[0] * 100\nmissing_tab = pd.DataFrame({'Count': count, 'Percent': pct})\nprint(missing_tab)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:21:00.676914Z","iopub.execute_input":"2022-08-08T10:21:00.677314Z","iopub.status.idle":"2022-08-08T10:21:00.736672Z","shell.execute_reply.started":"2022-08-08T10:21:00.677277Z","shell.execute_reply":"2022-08-08T10:21:00.735695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one-hot encoding of categorical features\ndf_all = pd.get_dummies(df_all, columns = categorical_vars)\nprint(df_all.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:21:00.737699Z","iopub.execute_input":"2022-08-08T10:21:00.738055Z","iopub.status.idle":"2022-08-08T10:21:00.773924Z","shell.execute_reply.started":"2022-08-08T10:21:00.738021Z","shell.execute_reply":"2022-08-08T10:21:00.772957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# standardizing\nX = df_all[numeric_vars]\nX = PowerTransformer().fit_transform(X)\ndf_all[numeric_vars] = RobustScaler().fit_transform(X)\ndf_all[numeric_vars] = X\ndf_all.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:21:00.775383Z","iopub.execute_input":"2022-08-08T10:21:00.775738Z","iopub.status.idle":"2022-08-08T10:21:01.805407Z","shell.execute_reply.started":"2022-08-08T10:21:00.775705Z","shell.execute_reply":"2022-08-08T10:21:01.804240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# iterative imputer\nimp = IterativeImputer(max_iter = 10, random_state = 0)\ndf_all[[x for x in df_all.columns if x not in  ['type', 'failure']]] = imp.fit_transform(df_all[[x for x in df_all.columns if x not in  ['type', 'failure']]])\ndf_all.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:21:01.806824Z","iopub.execute_input":"2022-08-08T10:21:01.807175Z","iopub.status.idle":"2022-08-08T10:22:31.157383Z","shell.execute_reply.started":"2022-08-08T10:21:01.807141Z","shell.execute_reply":"2022-08-08T10:22:31.156084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# no more missing values left\ncount = df_all.isna().sum()[df_all.isna().sum() > 0]\npct = df_all.isna().sum()[df_all.isna().sum() > 0] / df_all.shape[0] * 100\nmissing_tab = pd.DataFrame({'Count': count, 'Percent': pct})\nprint(missing_tab)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:22:31.158937Z","iopub.execute_input":"2022-08-08T10:22:31.160126Z","iopub.status.idle":"2022-08-08T10:22:31.206653Z","shell.execute_reply.started":"2022-08-08T10:22:31.160074Z","shell.execute_reply":"2022-08-08T10:22:31.205389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all[numeric_vars].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:22:31.208326Z","iopub.execute_input":"2022-08-08T10:22:31.209522Z","iopub.status.idle":"2022-08-08T10:22:31.339407Z","shell.execute_reply.started":"2022-08-08T10:22:31.209471Z","shell.execute_reply":"2022-08-08T10:22:31.336471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# divide preprocessed data back to train and test set\ndf_train = df_all.loc[df_all['type'] == 'train']\ndf_test = df_all.loc[df_all['type'] == 'test']\ndf_train.drop('type', axis = 1, inplace = True)\ndf_test.drop('type', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:22:31.341237Z","iopub.execute_input":"2022-08-08T10:22:31.341921Z","iopub.status.idle":"2022-08-08T10:22:31.371493Z","shell.execute_reply.started":"2022-08-08T10:22:31.341867Z","shell.execute_reply":"2022-08-08T10:22:31.370300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df_train['failure']\nX = df_train.loc[:, df_train.columns != 'failure']\n\n# Splitting the train data into train and validation set\ntrain_X, val_X, train_y, val_y = train_test_split(X, y, test_size = 0.30, random_state = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:22:31.373014Z","iopub.execute_input":"2022-08-08T10:22:31.373692Z","iopub.status.idle":"2022-08-08T10:22:31.395685Z","shell.execute_reply.started":"2022-08-08T10:22:31.373644Z","shell.execute_reply":"2022-08-08T10:22:31.394256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\naccuracy = {}\nfrom sklearn.metrics import accuracy_score\n\n# Logistic Regression with optimal parameters\nfrom sklearn.linear_model import LogisticRegression\nmodels['Logistic Regression'] = LogisticRegression(max_iter = 3000, penalty = 'l2', C = 0.0001)\n\n# Support Vector Machines\nfrom sklearn import svm\nmodels['Support Vector Machines'] = svm.SVC(kernel = 'poly', degree = 3)\n\n# Decision Trees\nfrom sklearn.tree import DecisionTreeClassifier\nmodels['Decision Trees'] = DecisionTreeClassifier()\n\n# Random Forest\nfrom sklearn.ensemble import RandomForestClassifier\nmodels['Random Forest'] = RandomForestClassifier()\n\n# Naive Bayes\nfrom sklearn.naive_bayes import GaussianNB\nmodels['Naive Bayes'] = GaussianNB()\n\n# K-Nearest Neighbors\nfrom sklearn.neighbors import KNeighborsClassifier\nmodels['K-Nearest Neighbor'] = KNeighborsClassifier()\nfrom sklearn.metrics import accuracy_score\n\n# XGBoost\nfrom xgboost.sklearn import XGBClassifier\nmodels['XGBoost'] = XGBClassifier()\n\n# LightGBM\nfrom lightgbm import LGBMClassifier\nmodels['LightGBM'] = LGBMClassifier()\n\n# CatBoost with optimal parameters\nfrom catboost import CatBoostClassifier\nparameters = {'depth'         : [4,5,6,7,8,9, 10],\n                 'learning_rate' : [0.01,0.02,0.03,0.04],\n                  'iterations'    : [10, 20,30,40,50,60,70,80,90, 100]\n                 }\nGrid_CBC = GridSearchCV(estimator = CatBoostClassifier(), param_grid = parameters, cv = 2, n_jobs=-1)\n# print(\" Results from Grid Search \" )\n# print(\"\\n The best estimator across ALL searched params:\\n\",Grid_CBC.best_estimator_)\n# print(\"\\n The best score across ALL searched params:\\n\",Grid_CBC.best_score_)\n# print(\"\\n The best parameters across ALL searched params:\\n\",Grid_CBC.best_params_)\n# CatBoost model with optimal parameters:\nGrid_CBC.fit(train_X, train_y, verbose = False)\npredictions = Grid_CBC.predict(val_X)\naccuracy['CatBoost'] = accuracy_score(predictions, val_y)\n\nfor key in models.keys():\n    models[key].fit(train_X, train_y)\n    predictions = models[key].predict(val_X)\n    accuracy[key] = accuracy_score(predictions, val_y)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NEURAL NETWORK\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras import optimizers\n\n# hyperparameters\nhidden_units = 50\nlearning_rate = 0.01\nhidden_layer_act = 'tanh'\noutput_layer_act = 'sigmoid'\nno_epochs = 100\n\nmodel = Sequential()\nmodel.add(Dense(hidden_units, input_dim = train_X.shape[1], activation = hidden_layer_act))\nmodel.add(Dense(hidden_units, activation = hidden_layer_act))\nmodel.add(Dense(1, activation = output_layer_act))\nsgd = tf.keras.optimizers.SGD(learning_rate = learning_rate)\nmodel.compile(loss = 'binary_crossentropy', optimizer = sgd, metrics = ['acc'])\n# model.summary()\n\nmodel.fit(train_X, train_y, epochs = no_epochs,  batch_size = 100, verbose = False)\n\n# predictions\npredictions = model.predict(val_X)\nrounded = [int(round(x[0])) for x in predictions]\n\nmodels['Neural Network'] = np.nan\naccuracy['Neural Network'] = accuracy_score(rounded, val_y)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T07:01:31.877891Z","iopub.status.idle":"2022-08-08T07:01:31.879167Z","shell.execute_reply.started":"2022-08-08T07:01:31.878838Z","shell.execute_reply":"2022-08-08T07:01:31.878867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_model = pd.DataFrame(index = accuracy.keys(), columns = ['Accuracy'])\ndf_model['Accuracy'] = accuracy.values()\ndf_model.Accuracy.sort_values(ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T07:01:31.880981Z","iopub.status.idle":"2022-08-08T07:01:31.881921Z","shell.execute_reply.started":"2022-08-08T07:01:31.881617Z","shell.execute_reply":"2022-08-08T07:01:31.881646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# FINAL MODEL - Logistic regression with optimal parameters, score = 0.58633\n# fitting model on the whole training set and prediction on test data\nfinal_model_variables = list(train_X.columns)\ntrain_X = df_train[final_model_variables]\ntrain_y = df_train['failure']\ntest_X = df_test[final_model_variables]\n\n# final model1\nmodel_final1 = LogisticRegression(max_iter = 3000, penalty = 'l2', C = 0.0001)\nmodel_final1.fit(train_X, train_y)\n\n# final prediction on test set\ny_pred_final1 = model_final1.predict_proba(test_X)[:,1]\n\n# saving results to df for submission\nId = df_test.index\ndata_tuples = list(zip(Id, y_pred_final1))\nsubmission_df = pd.DataFrame(data_tuples, columns = ['id','failure'])\nsubmission_df.to_csv('final_submission12.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T07:01:31.883633Z","iopub.status.idle":"2022-08-08T07:01:31.884543Z","shell.execute_reply.started":"2022-08-08T07:01:31.884225Z","shell.execute_reply":"2022-08-08T07:01:31.884254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FINAL MODEL - SVM, score = 0.49356\n# fitting model on the whole training set and prediction on test data\nfinal_model_variables = list(train_X.columns)\ntrain_X = df_train[final_model_variables]\ntrain_y = df_train['failure']\ntest_X = df_test[final_model_variables]\n\n# final model2\nmodel_final2 = svm.SVC(kernel = 'poly', degree = 2, probability = True)\nmodel_final2.fit(train_X, train_y)\n\n# final prediction on test set\ny_pred_final2 = model_final2.predict_proba(test_X)[:,1]\n\n# saving results to df for submission\nId = df_test.index\ndata_tuples = list(zip(Id, y_pred_final2))\nsubmission_df = pd.DataFrame(data_tuples, columns = ['id','failure'])\nsubmission_df.to_csv('final_submission2.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T07:01:31.886194Z","iopub.status.idle":"2022-08-08T07:01:31.887100Z","shell.execute_reply.started":"2022-08-08T07:01:31.886789Z","shell.execute_reply":"2022-08-08T07:01:31.886817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FINAL MODEL - neural network, score = 0.57070\n\n# hyperparameters\nhidden_units = 50\nlearning_rate = 0.01\nhidden_layer_act = 'tanh'\noutput_layer_act = 'sigmoid'\nno_epochs = 100\n\n\n# fitting model on the whole training set and prediction on test data\nfinal_model_variables = list(train_X.columns)\ntrain_X = df_train[final_model_variables]\ntrain_y = df_train['failure']\ntest_X = df_test[final_model_variables]\n\n# final model3\nmodel = Sequential()\nmodel.add(Dense(hidden_units, input_dim = train_X.shape[1], activation = hidden_layer_act))\nmodel.add(Dense(hidden_units, activation = hidden_layer_act))\nmodel.add(Dense(1, activation = output_layer_act))\nsgd = tf.keras.optimizers.SGD(learning_rate = learning_rate)\nmodel.compile(loss = 'binary_crossentropy',optimizer = 'adam', metrics = ['accuracy'])\nmodel.fit(train_X, train_y, epochs = no_epochs,  batch_size = 10, verbose = False)\n\n# final prediction on test set\ny_pred_final3 = model.predict(test_X)\ny_pred_final3 = pd.Series((chain(*y_pred_final3)))\n\n# saving results to df for submission\nId = df_test.index\ndata_tuples = list(zip(Id, y_pred_final3))\nsubmission_df = pd.DataFrame(data_tuples, columns = ['id','failure'])\nsubmission_df.to_csv('final_submission33.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T07:01:31.888769Z","iopub.status.idle":"2022-08-08T07:01:31.889669Z","shell.execute_reply.started":"2022-08-08T07:01:31.889361Z","shell.execute_reply":"2022-08-08T07:01:31.889405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FINAL MODEL - CatBoost with optimal parameters, score = 0.57002\n# fitting model on the whole training set and prediction on test data\nfinal_model_variables = list(train_X.columns)\ntrain_X = df_train[final_model_variables]\ntrain_y = df_train['failure']\ntest_X = df_test[final_model_variables]\n\n# final model4\nmodel_final4 = CatBoostClassifier(depth = 4, iterations = 10, learning_rate = 0.01)\nmodel_final4.fit(train_X, train_y, verbose = False)\n\n# final prediction on test set\ny_pred_final4 = model_final4.predict_proba(test_X)[:,1]\n\n# saving results to df for submission\nId = df_test.index\ndata_tuples = list(zip(Id, y_pred_final4))\nsubmission_df = pd.DataFrame(data_tuples, columns = ['id','failure'])\nsubmission_df.to_csv('final_submission44.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T07:01:31.891326Z","iopub.status.idle":"2022-08-08T07:01:31.892194Z","shell.execute_reply.started":"2022-08-08T07:01:31.891949Z","shell.execute_reply":"2022-08-08T07:01:31.891974Z"},"trusted":true},"execution_count":null,"outputs":[]}]}