{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import libraries\nimport numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport statistics \nimport scipy.stats \nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nimport statsmodels.api as sm\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import roc_curve\n%matplotlib inline\nsns.set_theme(style = \"whitegrid\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-20T09:07:15.433934Z","iopub.execute_input":"2022-07-20T09:07:15.434374Z","iopub.status.idle":"2022-07-20T09:07:15.442909Z","shell.execute_reply.started":"2022-07-20T09:07:15.434338Z","shell.execute_reply":"2022-07-20T09:07:15.441916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load train and test data\ntrain_data = pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\") \ntest_data = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\ntrain_data['type'] = 'train'\ntest_data['type'] = 'test'\nall_data = pd.concat([train_data, test_data], axis = 0, ignore_index = True)\nall_data.set_index('PassengerId')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:10:24.840749Z","iopub.execute_input":"2022-07-20T09:10:24.841198Z","iopub.status.idle":"2022-07-20T09:10:24.946847Z","shell.execute_reply.started":"2022-07-20T09:10:24.841163Z","shell.execute_reply":"2022-07-20T09:10:24.945536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# parse Cabin variable\nall_data[['Cabin1', 'Cabin2', 'Cabin3']] = all_data['Cabin'].str.split('/', expand = True)\n\n# drop unimportant variables\nall_data.drop(['Cabin', 'Name', 'Cabin2'], axis = 1, inplace = True)\nall_data.head()","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:29.244524Z","iopub.execute_input":"2022-07-20T09:10:29.245254Z","iopub.status.idle":"2022-07-20T09:10:29.304457Z","shell.execute_reply.started":"2022-07-20T09:10:29.245190Z","shell.execute_reply":"2022-07-20T09:10:29.303293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make list of numeric and categorical variables\nall_data.dtypes\nnumeric_vars = [i for i in all_data.columns if all_data.dtypes[i] != 'object']\ncategorical_vars = [i for i in all_data.columns if all_data.dtypes[i] == 'object']\ncategorical_vars.remove('PassengerId')\ncategorical_vars.remove('type')","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:32.477719Z","iopub.execute_input":"2022-07-20T09:10:32.478104Z","iopub.status.idle":"2022-07-20T09:10:32.487384Z","shell.execute_reply.started":"2022-07-20T09:10:32.478073Z","shell.execute_reply":"2022-07-20T09:10:32.486414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data[numeric_vars].describe() # numerical values are in reasonable range","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-19T21:40:34.975778Z","iopub.execute_input":"2022-07-19T21:40:34.976116Z","iopub.status.idle":"2022-07-19T21:40:35.013882Z","shell.execute_reply.started":"2022-07-19T21:40:34.976083Z","shell.execute_reply":"2022-07-19T21:40:35.013103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check all unique values of categorical variables\nfor i in categorical_vars:\n    print(i, ': ', all_data[i].unique()) ","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-19T21:40:35.017580Z","iopub.execute_input":"2022-07-19T21:40:35.018111Z","iopub.status.idle":"2022-07-19T21:40:35.032662Z","shell.execute_reply.started":"2022-07-19T21:40:35.018078Z","shell.execute_reply":"2022-07-19T21:40:35.031410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count and percent of missing values\ncount = all_data.isna().sum()[all_data.isna().sum() > 0]\npct = all_data.isna().sum()[all_data.isna().sum() > 0] / all_data.shape[0] * 100\nmissing_tab = pd.DataFrame({'Count': count, 'Percent': pct})\nprint(missing_tab)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:36.858414Z","iopub.execute_input":"2022-07-20T09:10:36.858830Z","iopub.status.idle":"2022-07-20T09:10:36.928632Z","shell.execute_reply.started":"2022-07-20T09:10:36.858797Z","shell.execute_reply":"2022-07-20T09:10:36.927394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# distribution of numerical features \ndf = pd.melt(all_data, value_vars = [x for x in numeric_vars])\ng = sns.FacetGrid(df, col = \"variable\", col_wrap = 3, sharey = False, sharex = False)\ng = g.map(sns.histplot, \"value\", bins = 20)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-19T21:40:35.122945Z","iopub.execute_input":"2022-07-19T21:40:35.123274Z","iopub.status.idle":"2022-07-19T21:40:36.432160Z","shell.execute_reply.started":"2022-07-19T21:40:35.123244Z","shell.execute_reply":"2022-07-19T21:40:36.431190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputing numerical missing values\nimputer_columns = [\"Age\", \"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\"]\nimputer = SimpleImputer(strategy = 'median')\nimputer.fit(all_data[imputer_columns])\nall_data[imputer_columns] = imputer.transform(all_data[imputer_columns])","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:41.629772Z","iopub.execute_input":"2022-07-20T09:10:41.630174Z","iopub.status.idle":"2022-07-20T09:10:41.651742Z","shell.execute_reply.started":"2022-07-20T09:10:41.630143Z","shell.execute_reply":"2022-07-20T09:10:41.650846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(all_data['HomePlanet'], all_data['Destination'])","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-19T21:40:36.455794Z","iopub.execute_input":"2022-07-19T21:40:36.456105Z","iopub.status.idle":"2022-07-19T21:40:36.479468Z","shell.execute_reply.started":"2022-07-19T21:40:36.456077Z","shell.execute_reply":"2022-07-19T21:40:36.478736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputing Destination and HomePlanet\nall_data.loc[all_data['Destination'].isnull(), 'Destination'] = 'TRAPPIST-1e'\nall_data.loc[(all_data['Destination'] == '55 Cancri e') & all_data['HomePlanet'].isnull(), 'HomePlanet'] = 'Europa'\nall_data.loc[(all_data['Destination'] == 'PSO J318.5-22') & all_data['HomePlanet'].isnull(), 'HomePlanet'] = 'Earth'\nall_data.loc[(all_data['Destination'] == 'TRAPPIST-1e') & all_data['HomePlanet'].isnull(), 'HomePlanet'] = 'Earth'\n\nall_data['HomePlanet'].fillna(statistics.mode(all_data['HomePlanet']), inplace = True)\nall_data['Destination'].fillna(statistics.mode(all_data['Destination']), inplace = True)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:44.982913Z","iopub.execute_input":"2022-07-20T09:10:44.983362Z","iopub.status.idle":"2022-07-20T09:10:45.020628Z","shell.execute_reply.started":"2022-07-20T09:10:44.983327Z","shell.execute_reply":"2022-07-20T09:10:45.019444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputing CryoSleep and converting coding False/True to 0/1\nall_data['CryoSleep'].fillna(statistics.mode(all_data['CryoSleep']), inplace = True)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:49.414068Z","iopub.execute_input":"2022-07-20T09:10:49.415161Z","iopub.status.idle":"2022-07-20T09:10:49.426215Z","shell.execute_reply.started":"2022-07-20T09:10:49.415117Z","shell.execute_reply":"2022-07-20T09:10:49.425310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputing VIP and converting coding False/True to 0/1\nall_data['VIP'].fillna(statistics.mode(all_data['VIP']), inplace = True)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:51.190917Z","iopub.execute_input":"2022-07-20T09:10:51.191358Z","iopub.status.idle":"2022-07-20T09:10:51.205248Z","shell.execute_reply.started":"2022-07-20T09:10:51.191325Z","shell.execute_reply":"2022-07-20T09:10:51.203982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imputing Cabin1 and Cabin3\nall_data['Cabin1'].fillna(statistics.mode(all_data['Cabin1']), inplace = True)\nall_data['Cabin3'].fillna(statistics.mode(all_data['Cabin3']), inplace = True)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:52.980523Z","iopub.execute_input":"2022-07-20T09:10:52.980939Z","iopub.status.idle":"2022-07-20T09:10:52.997868Z","shell.execute_reply.started":"2022-07-20T09:10:52.980906Z","shell.execute_reply":"2022-07-20T09:10:52.996597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count and percent of missing values, no missing values left\ncount = all_data.isna().sum()[all_data.isna().sum() > 0]\npct = all_data.isna().sum()[all_data.isna().sum() > 0] / all_data.shape[0] * 100\nmissing_tab = pd.DataFrame({'Count': count, 'Percent': pct})\nprint(missing_tab)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:10:56.578854Z","iopub.execute_input":"2022-07-20T09:10:56.580110Z","iopub.status.idle":"2022-07-20T09:10:56.638943Z","shell.execute_reply.started":"2022-07-20T09:10:56.580051Z","shell.execute_reply":"2022-07-20T09:10:56.637720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# EDA, impact of variables on target\nsns.histplot(all_data[\"Age\"][all_data.Transported == True], color=\"darkturquoise\")\nsns.histplot(all_data[\"Age\"][all_data.Transported == False], color=\"lightcoral\")\nplt.legend(['Transported', 'Not transported'])\nplt.show() # seems like Age won't have an impact on target","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-19T21:40:36.640275Z","iopub.execute_input":"2022-07-19T21:40:36.640949Z","iopub.status.idle":"2022-07-19T21:40:37.043269Z","shell.execute_reply.started":"2022-07-19T21:40:36.640918Z","shell.execute_reply":"2022-07-19T21:40:37.042228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# chi square test of independence for all pairs of categorical features\nCramersV = []\nfor i in categorical_vars:\n    for j in categorical_vars:\n        if i != j:\n            ct = pd.crosstab(index = all_data[i], columns = all_data[j])\n            stat, p, dof, expected = scipy.stats.chi2_contingency(ct)\n            n = np.sum(np.sum(ct))\n            minDim = min(ct.shape) - 1\n            \n            # calculate Cramer's V \n            V = (np.sqrt((stat/n) / minDim))\n            CramersV.append([i, j, V])\n            \n# table of the values of CramersV for all pairs of categorical features\nCramersV = pd.DataFrame(CramersV)\nCramersV.rename(columns = {0: \"Feature1\", 1: \"Feature2\", 2: \"Cramers_V\"}, inplace = True)\nCramersV.sort_values('Cramers_V', ascending = False, inplace = True)\nCramersV = CramersV.iloc[::2,:]\nCramersV.head(10)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-19T21:40:37.046894Z","iopub.execute_input":"2022-07-19T21:40:37.047602Z","iopub.status.idle":"2022-07-19T21:40:37.561260Z","shell.execute_reply.started":"2022-07-19T21:40:37.047557Z","shell.execute_reply":"2022-07-19T21:40:37.560136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:11:03.564902Z","iopub.execute_input":"2022-07-20T09:11:03.565356Z","iopub.status.idle":"2022-07-20T09:11:03.590474Z","shell.execute_reply.started":"2022-07-20T09:11:03.565319Z","shell.execute_reply":"2022-07-20T09:11:03.589286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one-hot encoding of categorical features\nencoding_cols = [x for x in categorical_vars if x != 'Transported']\nall_data = pd.get_dummies(all_data, columns = encoding_cols)\nprint(all_data.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:11:08.797838Z","iopub.execute_input":"2022-07-20T09:11:08.798764Z","iopub.status.idle":"2022-07-20T09:11:08.824606Z","shell.execute_reply.started":"2022-07-20T09:11:08.798707Z","shell.execute_reply":"2022-07-20T09:11:08.823218Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# standardizing numeric columns\nfrom sklearn.preprocessing import StandardScaler\nX = all_data[numeric_vars]\nstandardizer = StandardScaler()\nX = standardizer.fit_transform(X)\nall_data[numeric_vars] = X\nall_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:11:15.350480Z","iopub.execute_input":"2022-07-20T09:11:15.350886Z","iopub.status.idle":"2022-07-20T09:11:15.383031Z","shell.execute_reply.started":"2022-07-20T09:11:15.350852Z","shell.execute_reply":"2022-07-20T09:11:15.381919Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# divide preprocessed data back to train and test set\ntrain_data = all_data.loc[all_data['type'] == 'train']\ntest_data = all_data.loc[all_data['type'] == 'test']\ntrain_data.drop(['type', 'PassengerId'], axis = 1, inplace = True)\ntrain_data['Transported'] = [int(x) for x in train_data['Transported']] # converting target to 0 and 1\n\ntest_data.drop(['type', 'PassengerId'], axis = 1, inplace = True)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:11:30.493209Z","iopub.execute_input":"2022-07-20T09:11:30.493646Z","iopub.status.idle":"2022-07-20T09:11:30.521759Z","shell.execute_reply.started":"2022-07-20T09:11:30.493614Z","shell.execute_reply":"2022-07-20T09:11:30.520931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_data['Transported']\nX = train_data.loc[:, train_data.columns != 'Transported']\n\n# Splitting the train data into train and validation set\ntrain_X, val_X, train_y, val_y = train_test_split(X, y, test_size = 0.30, random_state = 1)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-20T09:15:22.843867Z","iopub.execute_input":"2022-07-20T09:15:22.844357Z","iopub.status.idle":"2022-07-20T09:15:22.856327Z","shell.execute_reply.started":"2022-07-20T09:15:22.844320Z","shell.execute_reply":"2022-07-20T09:15:22.855203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\n\n# Logistic Regression\nfrom sklearn.linear_model import LogisticRegression\nmodels['Logistic Regression'] = LogisticRegression(max_iter = 300)\n\n# Support Vector Machines\nfrom sklearn import svm\nmodels['Support Vector Machines'] = svm.SVC(kernel = 'poly', degree = 2)\n\n# Decision Trees\nfrom sklearn.tree import DecisionTreeClassifier\nmodels['Decision Trees'] = DecisionTreeClassifier()\n\n# Random Forest\nfrom sklearn.ensemble import RandomForestClassifier\nmodels['Random Forest'] = RandomForestClassifier()\n\n# Naive Bayes\nfrom sklearn.naive_bayes import GaussianNB\nmodels['Naive Bayes'] = GaussianNB()\n\n# K-Nearest Neighbors\nfrom sklearn.neighbors import KNeighborsClassifier\nmodels['K-Nearest Neighbor'] = KNeighborsClassifier()\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score\n\naccuracy, precision, recall = {}, {}, {}\n\nfor key in models.keys():\n    \n    # Fit the classifier model\n    models[key].fit(train_X, train_y)\n    \n    # Prediction on validation data\n    predictions = models[key].predict(val_X)\n    \n    # Calculate Accuracy, Precision and Recall Metrics\n    accuracy[key] = accuracy_score(predictions, val_y)\n    precision[key] = precision_score(predictions, val_y)\n    recall[key] = recall_score(predictions, val_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:16:24.446194Z","iopub.execute_input":"2022-07-20T09:16:24.447427Z","iopub.status.idle":"2022-07-20T09:16:27.867943Z","shell.execute_reply.started":"2022-07-20T09:16:24.447380Z","shell.execute_reply":"2022-07-20T09:16:27.866737Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NEURAL NETWORK\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras import optimizers\n\n# hyperparameters\nhidden_units = 50\nlearning_rate = 0.01\nhidden_layer_act = 'tanh'\noutput_layer_act = 'sigmoid'\nno_epochs = 100\n\nmodel = Sequential()\nmodel.add(Dense(hidden_units, input_dim = train_X.shape[1], activation = hidden_layer_act))\nmodel.add(Dense(hidden_units, activation = hidden_layer_act))\nmodel.add(Dense(1, activation = output_layer_act))\nsgd = tf.keras.optimizers.SGD(learning_rate = learning_rate)\nmodel.compile(loss = 'binary_crossentropy',optimizer = sgd, metrics = ['acc'])\n# model.summary()\n\nmodel.fit(train_X, train_y, epochs = no_epochs,  batch_size = 100, verbose = False)\n\n# predictions\npredictions = model.predict(val_X)\nrounded = [int(round(x[0])) for x in predictions]\n\nmodels['Neural Network'] = np.nan\naccuracy['Neural Network'] = accuracy_score(rounded, val_y)\nprecision['Neural Network'] = precision_score(rounded, val_y)\nrecall['Neural Network'] = recall_score(rounded, val_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:16:57.389351Z","iopub.execute_input":"2022-07-20T09:16:57.389851Z","iopub.status.idle":"2022-07-20T09:17:11.115083Z","shell.execute_reply.started":"2022-07-20T09:16:57.389807Z","shell.execute_reply":"2022-07-20T09:17:11.113619Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_model = pd.DataFrame(index = models.keys(), columns = ['Accuracy', 'Precision', 'Recall'])\ndf_model['Accuracy'] = accuracy.values()\ndf_model['Precision'] = precision.values()\ndf_model['Recall'] = recall.values()\n\ndf_model","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:17:11.117635Z","iopub.execute_input":"2022-07-20T09:17:11.118092Z","iopub.status.idle":"2022-07-20T09:17:11.138021Z","shell.execute_reply.started":"2022-07-20T09:17:11.118053Z","shell.execute_reply":"2022-07-20T09:17:11.136583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FINAL MODEL - logistic regression\n# fitting model on the whole training set and prediction on test data\nfinal_model_variables = list(train_X.columns)\ntrain_X = train_data[final_model_variables]\ntrain_y = train_data['Transported']\n\ntest_X = test_data[final_model_variables]\n\n# final model1\nmodel_final1 = LogisticRegression(max_iter = 3000)\nmodel_final1.fit(train_X, train_y)\n\n# final prediction on test set\ny_pred_final1 = model_final1.predict(test_X)\n\n# saving results to df for submission\nPassengerId = list(all_data.loc[all_data['type'] == 'test', 'PassengerId'])\nTransported = [bool(x) for x in y_pred_final1]\ndata_tuples = list(zip(PassengerId, Transported))\nsubmission_df = pd.DataFrame(data_tuples, columns = ['PassengerId','Transported'])\nsubmission_df.to_csv('final_submission1.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:17:21.321314Z","iopub.execute_input":"2022-07-20T09:17:21.321707Z","iopub.status.idle":"2022-07-20T09:17:21.495510Z","shell.execute_reply.started":"2022-07-20T09:17:21.321676Z","shell.execute_reply":"2022-07-20T09:17:21.494069Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FINAL MODEL - SVM\n# fitting model on the whole training set and prediction on test data\n\n# final model2\n# Support Vector Machines\nmodel_final2 = svm.SVC(kernel = 'poly', degree = 2)\nmodel_final2.fit(train_X, train_y)\n\n# final prediction on test set\ny_pred_final2 = model_final2.predict(test_X)\n\n# saving results to df for submission\nPassengerId = list(all_data.loc[all_data['type'] == 'test', 'PassengerId'])\nTransported = [bool(x) for x in y_pred_final2]\ndata_tuples = list(zip(PassengerId, Transported))\nsubmission_df = pd.DataFrame(data_tuples, columns = ['PassengerId','Transported'])\nsubmission_df.to_csv('final_submission2.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:17:26.451393Z","iopub.execute_input":"2022-07-20T09:17:26.451765Z","iopub.status.idle":"2022-07-20T09:17:30.553096Z","shell.execute_reply.started":"2022-07-20T09:17:26.451736Z","shell.execute_reply":"2022-07-20T09:17:30.551795Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FINAL MODEL - NEURAL NETWORK\n# fitting model on the whole training set and prediction on test data\n\n# final model3\n# neural network\nmodel.fit(train_X, train_y, epochs = no_epochs,  batch_size = 100, verbose = False)\n\n# final prediction on test set\npredictions = model.predict(test_X)\nrounded = [int(round(x[0])) for x in predictions]\n\n# saving results to df for submission\nPassengerId = list(all_data.loc[all_data['type'] == 'test', 'PassengerId'])\nTransported = [bool(x) for x in rounded]\ndata_tuples = list(zip(PassengerId, Transported))\nsubmission_df = pd.DataFrame(data_tuples, columns = ['PassengerId','Transported'])\nsubmission_df.to_csv('final_submission3.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:17:30.555177Z","iopub.execute_input":"2022-07-20T09:17:30.555588Z","iopub.status.idle":"2022-07-20T09:17:48.045626Z","shell.execute_reply.started":"2022-07-20T09:17:30.555555Z","shell.execute_reply":"2022-07-20T09:17:48.044688Z"},"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]}]}