{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\n\n#Pre-processing\nfrom sklearn import preprocessing\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nimport scipy.stats as stats\nfrom scipy.stats import chi2_contingency\n\n#Decomposition\nfrom sklearn.decomposition import PCA\n\n#Modeling \nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cluster\nfrom sklearn.metrics import accuracy_score\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-04T11:35:37.403431Z","iopub.execute_input":"2022-08-04T11:35:37.403975Z","iopub.status.idle":"2022-08-04T11:35:37.415611Z","shell.execute_reply.started":"2022-08-04T11:35:37.403930Z","shell.execute_reply":"2022-08-04T11:35:37.414683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Introduction\n### 1.1 | Table of contents\n\n### 1.2 | Data Loading and Cleaning\n* Initial Data Loading\n* Investigating Nan's\n* Nan Inputation with KMeans\n* Measuring Skewness and Kurtosis\n* Data Transformation\n\n### 1.3 | EDA of Numeric Variables\n* Visualizing distribution vs. response variable\n* Statistical Testing for predictor significance\n* Checking for Collinearity\n\n### 1.4 | EDA of Categorical Variables\n* Visualizing distribution vs. response variable\n* Statistical Testing for predictor significance\n\n### 1.5 | Categorical Encoding\n* Using dummy encoding to prepare data for modeling\n\n### 1.6 | EDA Conclusion\n\n### 1.7 | Model Training + Submission\n* Creating test/eval set\n* Fitting Logistic Regression Model\n* Submitting Results\n","metadata":{}},{"cell_type":"markdown","source":"## 1.2 | Data Loading and Cleaning\n\nStarting out we'll load the training set and take a brief look at the top 10 samples.","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:35:37.467192Z","iopub.execute_input":"2022-08-04T11:35:37.468225Z","iopub.status.idle":"2022-08-04T11:35:37.608513Z","shell.execute_reply.started":"2022-08-04T11:35:37.468177Z","shell.execute_reply":"2022-08-04T11:35:37.607668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on the above dataframe we can see that the majority of our predictor variables are numeric and continuous with the exception of attribute_0, attribute_1, and product code.","metadata":{}},{"cell_type":"markdown","source":"### Investigating NaN value's\n\nGraphing the NaN values below we can see an interesting feature with measurements_3 through 17 having a steadily inreasing number of NaN's.","metadata":{}},{"cell_type":"code","source":"#Graphing missing values\nnas = train.isnull().sum()\nplt.bar(nas.index, nas)\nplt.xticks(rotation = 90)\nplt.title('NA Count by Measurement')\nplt.show()\n\n#Printing D-types\n#print(df.dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:35:37.609891Z","iopub.execute_input":"2022-08-04T11:35:37.610344Z","iopub.status.idle":"2022-08-04T11:35:38.594590Z","shell.execute_reply.started":"2022-08-04T11:35:37.610316Z","shell.execute_reply":"2022-08-04T11:35:38.593581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### NaN Imputation with K-Means\n\nGiven the number of Nan values we need an accurate imputer to fill these in. Rather than using mean or median we'll use a K Nearest Neighbors imputer which will impute the missing values using the average of the n nearest neighbors.","metadata":{}},{"cell_type":"code","source":"warnings.filterwarnings('ignore')\n#Imputing missing values using a KNN imputer\nnumerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n\n#Breaking into numeric\nnumeric_df = train.select_dtypes(include=numerics)\nnumeric_df = numeric_df.drop(['failure'], axis = 1)\nnumeric_df_test = test.select_dtypes(include=numerics)\n\nmulti_imp = IterativeImputer(max_iter=9, random_state=42, verbose = 0,\n                            skip_complete = True, n_nearest_features = 10,\n                            tol = 0.001)\nmulti_imp.fit(numeric_df)\n\n#Imputing missing vals\nnumeric_df[:] = multi_imp.transform(numeric_df)\nnumeric_df_test[:] = multi_imp.transform(numeric_df_test)\n\n#Concatanating transformed numeric df back with categorical columns\ntrain = pd.concat([numeric_df, train[['product_code', 'attribute_0', 'attribute_1', 'failure']]], axis = 1)\ntest = pd.concat([numeric_df_test, test[['product_code', 'attribute_0', 'attribute_1']]], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:35:38.595875Z","iopub.execute_input":"2022-08-04T11:35:38.596180Z","iopub.status.idle":"2022-08-04T11:35:43.137942Z","shell.execute_reply.started":"2022-08-04T11:35:38.596153Z","shell.execute_reply":"2022-08-04T11:35:43.137079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Measuring Skewness and Kurtosis\n\nNext we'll examine the skewness and kurtosis of our predictor variables to determine if any data transformation is required.\n\nWe'll do this two ways, first by plotting a histrogram of each of the variables and next by calculating the skewness/Kurtosis values.","metadata":{}},{"cell_type":"code","source":"# Creating distribution plots\nfig, axs = plt.subplots(nrows=6, ncols=4)\naxs = axs.ravel()\nnumeric_cols = train.select_dtypes(include=numerics).columns\nplt.suptitle(\"Numeric Variable Distribution\", size=16)\nfor i in range(len(numeric_cols)):\n    axs[i].hist(train[numeric_cols[i]])\n    if train[numeric_cols[i]].dtype == 'float64':\n        axs[i].hist(train[numeric_cols[i]],  color = 'blue')\n        axs[i].get_xaxis().set_visible(False)\n        axs[i].get_yaxis().set_visible(False)\n    else:\n        axs[i].hist(train[numeric_cols[i]], color = 'red')\n        axs[i].get_xaxis().set_visible(False)\n        axs[i].get_yaxis().set_visible(False)\n        \n    axs[i].set_title(numeric_cols[i])\n    \nfig.set_size_inches(18.5, 10.5, forward=True)  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:35:43.140044Z","iopub.execute_input":"2022-08-04T11:35:43.140324Z","iopub.status.idle":"2022-08-04T11:35:44.609051Z","shell.execute_reply.started":"2022-08-04T11:35:43.140298Z","shell.execute_reply":"2022-08-04T11:35:44.608069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skewness = numeric_df.skew(axis = 0)\nkurtosis = numeric_df.kurt(axis = 0)\ninitial_assessment = pd.DataFrame(np.array([numeric_df.columns, skewness, kurtosis]).T,\n                                 columns = ['Variable', 'Skewness', 'Kurtosis'])\ninitial_assessment['Skew Clasification'] = np.where((abs(initial_assessment['Skewness']) < 0.5) , 'Symmetrical', 'Skewed')\ninitial_assessment.head(n=20)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:35:44.610273Z","iopub.execute_input":"2022-08-04T11:35:44.610546Z","iopub.status.idle":"2022-08-04T11:35:44.641005Z","shell.execute_reply.started":"2022-08-04T11:35:44.610520Z","shell.execute_reply":"2022-08-04T11:35:44.640094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Generally if the skewness is between -0.5 and 0.5, the data are considered to be fairly symmetrical, based on this standard we can see that 3 variables (loading, measurement_0, measurement_2) are skewed.","metadata":{}},{"cell_type":"markdown","source":"### Data Transformation\n\nBased on the above evaluation of kurtosis/skewness we can see that at least 4 of our variables are skewed and, this can cause issues for statistical tests and models that assume normally distributed samples. To address this we'll normalize the most skewed variables with a L2 normalization approach.","metadata":{}},{"cell_type":"code","source":"#Initial Data Cleaning\n#train = train.drop(['id'], axis = 1)\n#test = test.drop(['id'], axis = 1)\nattributes = ['attribute_2', 'attribute_3', 'measurement_4', 'measurement_5', 'measurement_6']\ntrain[attributes] = preprocessing.normalize(train[attributes])\ntest[attributes] = preprocessing.normalize(test[attributes])\ntrain.head(n=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:35:44.642496Z","iopub.execute_input":"2022-08-04T11:35:44.643025Z","iopub.status.idle":"2022-08-04T11:35:44.683230Z","shell.execute_reply.started":"2022-08-04T11:35:44.642988Z","shell.execute_reply":"2022-08-04T11:35:44.682233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3| EDA of Numeric Variables\n\nWe'll now use pairlplots to visualize the distribution of each of our numeric variables grouped by failure(either 0 or 1).  This will allow us to visually evaluate wheter the distribution is different for samples that failed versus succeeded.","metadata":{}},{"cell_type":"code","source":"#Now we have a normalized dataset without NaN values, creating pairsplot\nfirst_half = ['measurement_0','measurement_1','measurement_2','measurement_3',\n              'measurement_4','measurement_5','measurement_6','measurement_7','failure']\nsecond_half = ['measurement_8','measurement_9','measurement_10','measurement_11',\n              'measurement_12','measurement_13','measurement_14','measurement_15','failure']\n\nsns.set(rc = {'figure.figsize':(15,12)})\ng1 = sns.pairplot(train[first_half].sample(n=100),hue = 'failure', \n                 corner = True)\ng1.fig.suptitle(\"Correlation Between Measurements 0-7 and failure\", y = 1.01)\n\nsns.set(rc = {'figure.figsize':(15,12)})\ng2 = sns.pairplot(train[second_half].sample(n=100),hue = 'failure', \n                 corner = True)\ng2.fig.suptitle(\"Correlation Between Measurements 8 - 15 and failure\", y = 1.01)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:35:44.684877Z","iopub.execute_input":"2022-08-04T11:35:44.685233Z","iopub.status.idle":"2022-08-04T11:36:02.549167Z","shell.execute_reply.started":"2022-08-04T11:35:44.685197Z","shell.execute_reply":"2022-08-04T11:36:02.548227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Two Sample T-test\n\nWhile we can clearly see thta some of the variable distributions differ for failed versus suceded samples(Such as measurement_3, 0, and 2) we can get a more robust andswer to wether variable distributions differ based on our categorical outcome with a two sample t-test.\n\nA two Sample T-test can be used to test whether two populations have the same mean, in this case if we find that when we subset a measurement by failure(0 or 1) the means are different than we can conclude that variable likely has an impact on failure and will have some predictive power when used in a model.","metadata":{}},{"cell_type":"code","source":"cols_of_interest = train.iloc[:,5:21].columns\np_vals = []\nfor i in cols_of_interest:\n    group_1 = train[train.failure == 0][i]\n    group_2 = train[train.failure == 1][i]\n    statistic, p = stats.ttest_ind(a=group_1, b=group_2, equal_var=False)\n    p_vals.append(p)\n    \np_summary = pd.DataFrame(np.array([cols_of_interest, p_vals]).T, columns = ['Variable', 'p_value']).sort_values(by=['p_value'], ascending  =True)\n#p_summary = p_summary.sort_values(by=['p_value'], ascending  =True)\np_summary.head(n=10)\n\nfig = plt.figure(constrained_layout = True, figsize = [10,10])\nax1 = fig.add_subplot(111)\nsns.barplot(data = p_summary, x = 'Variable', y = 'p_value').set(title = 'p-values, variable impact on success/failure')\nplt.xticks(rotation=90)\nax1.axhline(0.05)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:36:02.550654Z","iopub.execute_input":"2022-08-04T11:36:02.550946Z","iopub.status.idle":"2022-08-04T11:36:03.126170Z","shell.execute_reply.started":"2022-08-04T11:36:02.550921Z","shell.execute_reply":"2022-08-04T11:36:03.125017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on our two sample t test we can see that there are 6 variables that have a p-value of less than 6: Measurements 17, 8, 4, 7, 2, 5. \n\nWhile were not going to immediately discard the rest of them it will be worth looking into dropping some of them during the model training phase and see if this increases our accuracy. Especially for p_values above 0.5 we may be able to reduce overfitting by dropping some of these values.","metadata":{}},{"cell_type":"markdown","source":"## Checking for Collinearity\n\nOne final test we'll perform on our numeric data is checking for collinearity; if multiple predictor values are closely correlated with each other this has the potential to increase uncertainty of our coefficient estimates and, more importantly in this application, can resulting in overfitting by introducing multiple variables that encode highly similar information. More here:https://en.wikipedia.org/wiki/Collinearity\n\nTo check for this we'll create a basic heatmap of our numeric variables.","metadata":{}},{"cell_type":"code","source":"num_corr = train[cols_of_interest].corr(method = 'pearson')\nax = sns.heatmap(num_corr, annot=True, cmap=\"YlGnBu\", linewidths=.5, fmt = '.2f').set(title = 'Pearson Correlation of Numeric Variables')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:36:03.127572Z","iopub.execute_input":"2022-08-04T11:36:03.127886Z","iopub.status.idle":"2022-08-04T11:36:04.444669Z","shell.execute_reply.started":"2022-08-04T11:36:03.127858Z","shell.execute_reply":"2022-08-04T11:36:04.443555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Above we can see that there are two pairs fo variables with correlation coefficients of magnitude greater than 0.5; Measurement_8/Meassurement_17 and Measurement_5/Measurement_6. Based on this we can try eliminating one of each of these variables during model training and see if it impacts the results. ","metadata":{}},{"cell_type":"markdown","source":"## Categorical Variables\n\nNow we'll examine the importance of our categorical variables both visually and statistically to see if they have any impact on product failure.\n","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3)\n\ng3 = sns.countplot( ax=ax[0], data = train, x = 'attribute_0',  hue = 'failure')\ng3.set(title = 'Attribute 0 versus failure')\n\ng4 = sns.countplot(ax=ax[1], data = train, x = 'attribute_1',  hue = 'failure')\ng4.set(title = 'Attribute 1 versus failure')\n\ng5 = sns.countplot(ax=ax[2], data = train, x = 'product_code',  hue = 'failure')\ng5.set(title = 'Product Code versus failure')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:36:04.448176Z","iopub.execute_input":"2022-08-04T11:36:04.448528Z","iopub.status.idle":"2022-08-04T11:36:05.003514Z","shell.execute_reply.started":"2022-08-04T11:36:04.448492Z","shell.execute_reply":"2022-08-04T11:36:05.002406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on the above countplots its difficult to tell if attribute_0, attribute_1, or product code really impacts failure, to get a more exact measure we'll perform some statistical testing to get a more exact answer.","metadata":{}},{"cell_type":"markdown","source":"## Chi Square Test\n\nNow we'll test for correlation between our dependent and independent variables starting with our categorical ones: attribute_0 and attribute_1.\n\nWe'll use a Chi Square test to determine if there's a statistically significant difference in the expected versus observed distributions https://en.wikipedia.org/wiki/Chi-squared_test","metadata":{}},{"cell_type":"code","source":"#Creating contigency levels\ndata_crosstab1 = pd.crosstab(train['attribute_0'],\n                            train['failure'],\n                           margins=True, margins_name=\"Total\")\ndata_crosstab2 = pd.crosstab(train['attribute_1'],\n                            train['failure'],\n                           margins=True, margins_name=\"Total\")\ndata_crosstab3 = pd.crosstab(train['product_code'],\n                            train['failure'],\n                           margins=True, margins_name=\"Total\")\n\n#Calculating p values\nstat1, p1, dof1, expected1 = chi2_contingency(data_crosstab1)\nstat2, p2, dof2, expected2 = chi2_contingency(data_crosstab2)\nstat3, p3, dof3, expected3 = chi2_contingency(data_crosstab3)\nprint('P-Val, Attribute_0: ',p1)\nprint('P-Val, Attribute_1: ',p2)\nprint('P-Val, Product Code: ',p3)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:36:05.005325Z","iopub.execute_input":"2022-08-04T11:36:05.005714Z","iopub.status.idle":"2022-08-04T11:36:05.124283Z","shell.execute_reply.started":"2022-08-04T11:36:05.005684Z","shell.execute_reply":"2022-08-04T11:36:05.123175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This test returns p-value of 0.211, 0.609, and 0.22 for attribute_0, attribute_1, and product code respectively so at a significance level of 0.05 we would fail to reject the null hypothesis and conclude that there is no relation between variables attribute_1/attribute_2/product code and failure.\n\nThis aligns with our previous countplot which shows a roughly equal proportion of failure/sucess regardless of the attribute_0 or 1 level.","metadata":{}},{"cell_type":"markdown","source":"## 1.5 | Categorical Encoding\n\nPrior to model development we'll need to convert our categorical variables to a more friendly format. To do this we'll use the Pandas dummy encoder which will break each factor level into a separate binary variable. The typical concern with this approach is that it can cause overfitting by drastically increasing the number of variables but the maximum number of factor levels in our categorical varaibles is 5 in product code so this shouldnt be a significant concern here.","metadata":{}},{"cell_type":"code","source":"test = test.drop(['product_code'], axis = 1)\ntrain = train.drop(['product_code'], axis = 1)\nencoded_columns = ['attribute_0', 'attribute_1']\ntest.head(n=10)\nfor column in encoded_columns:\n    tempdf = pd.get_dummies(train[column], prefix=column)\n    tempdf_test = pd.get_dummies(test[column], prefix=column)\n    \n    train = pd.merge(\n        left=train,\n        right=tempdf,\n        left_index=True,\n        right_index=True,\n    )\n    test = pd.merge(\n        left=test,\n        right=tempdf_test,\n        left_index=True,\n        right_index=True,\n    )\ntrain = train.drop(encoded_columns, axis = 1)\ntest = test.drop(encoded_columns, axis = 1)\ntrain.head(n=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:36:05.125442Z","iopub.execute_input":"2022-08-04T11:36:05.126191Z","iopub.status.idle":"2022-08-04T11:36:05.183093Z","shell.execute_reply.started":"2022-08-04T11:36:05.126159Z","shell.execute_reply":"2022-08-04T11:36:05.182038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.7 | EDA Conclusions:\n\nBased on the previou EDA we can draw several conclusions that will help us in model development:\n1. First with regards to categorical data: None of our categorical variables are highly correlated with failure chance and we can try dropping them one at a time to see if this impacts our model accuracy\n* Will likely start with Product code since this had the largest p-value on our Chi Square test\n2. For numeric variables we were able to identify Measurement_8/Meassurement_17 and Measurement_5/Measurement_6 as significantly correlated, abs(pearson coefficient) > 0.5\n* Again we can try and drop one of each pair during training to see if we can reduce dimensionality, and overfitting, while maintaining accuracy\n","metadata":{}},{"cell_type":"markdown","source":"## 1.6 | Model Training + Submission\n\nFor this portion we'll use a logistic regression model to provide a baseline for layer modelign efforts and because logistic regression performs surprisingly well on this dataset.","metadata":{}},{"cell_type":"markdown","source":"### Test-Train Split","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(train.drop(['failure'], axis = 1), \n                                                    train['failure'], test_size=0.33, random_state=42)\nX_train.head(n=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:36:05.184336Z","iopub.execute_input":"2022-08-04T11:36:05.184924Z","iopub.status.idle":"2022-08-04T11:36:05.221999Z","shell.execute_reply.started":"2022-08-04T11:36:05.184894Z","shell.execute_reply":"2022-08-04T11:36:05.220973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Training","metadata":{}},{"cell_type":"code","source":"log_model = LogisticRegression(random_state=0)\nlog_model.fit(X_train, y_train)\n\nlog_pred = log_model.predict(X_test)\nlog_table = cluster.contingency_matrix(y_test, log_pred)\nprint(log_table)\naccuracy_score(y_test, log_pred)\n\nfig = plt.figure(num=None, figsize=(8, 6), dpi=80, facecolor='w', edgecolor='k')\nplt.clf()\nax = fig.add_subplot(111)\nax.set_aspect(1)\nres = sns.heatmap(log_table, annot=True, fmt='.2f', cmap=\"YlGnBu\", vmin=0.0, vmax=100.0)\nplt.title('Logistic Regression Contigency Table',fontsize=12)\nplt.xticks([i+0.5 for i in range(log_table.shape[0])], ['Success', 'Failure'])\nplt.xticks(rotation=0)\nplt.yticks([i+0.5 for i in range(log_table.shape[1])], ['Success', 'Failure'])\nplt.yticks(rotation=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:36:05.223607Z","iopub.execute_input":"2022-08-04T11:36:05.224286Z","iopub.status.idle":"2022-08-04T11:36:05.741296Z","shell.execute_reply.started":"2022-08-04T11:36:05.224231Z","shell.execute_reply":"2022-08-04T11:36:05.740481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submitting Results","metadata":{}},{"cell_type":"code","source":"log_prediction = log_model.predict_proba(test)[:,1]\n\nlabels = test['id']\nlog_submission = pd.DataFrame(np.array([labels, log_prediction]).T,\n                                 columns = ['id', 'failure'])\nlog_submission['id'] = log_submission['id'].astype(int)\n\n#Submitting\nlog_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T11:36:05.742277Z","iopub.execute_input":"2022-08-04T11:36:05.743255Z","iopub.status.idle":"2022-08-04T11:36:05.885312Z","shell.execute_reply.started":"2022-08-04T11:36:05.743223Z","shell.execute_reply":"2022-08-04T11:36:05.884252Z"},"trusted":true},"execution_count":null,"outputs":[]}]}