{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn import preprocessing\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import accuracy_score\n\nfrom pandas_profiling import ProfileReport","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T06:15:42.235541Z","iopub.execute_input":"2022-08-02T06:15:42.235957Z","iopub.status.idle":"2022-08-02T06:15:43.912935Z","shell.execute_reply.started":"2022-08-02T06:15:42.235916Z","shell.execute_reply":"2022-08-02T06:15:43.911984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/spaceship-titanic/test.csv')\ntrain = pd.read_csv('../input/spaceship-titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:34.360276Z","iopub.execute_input":"2022-08-02T06:10:34.361169Z","iopub.status.idle":"2022-08-02T06:10:34.441589Z","shell.execute_reply.started":"2022-08-02T06:10:34.361121Z","shell.execute_reply":"2022-08-02T06:10:34.440466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:34.443146Z","iopub.execute_input":"2022-08-02T06:10:34.443771Z","iopub.status.idle":"2022-08-02T06:10:34.475086Z","shell.execute_reply.started":"2022-08-02T06:10:34.443725Z","shell.execute_reply":"2022-08-02T06:10:34.473843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ProfileReport(train, title='EDA Report Spaceship Titanic')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:16:20.832922Z","iopub.execute_input":"2022-08-02T06:16:20.833702Z","iopub.status.idle":"2022-08-02T06:16:39.412772Z","shell.execute_reply.started":"2022-08-02T06:16:20.833663Z","shell.execute_reply":"2022-08-02T06:16:39.411769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round(train.isna().sum()/len(train)*100,2)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:34.478208Z","iopub.execute_input":"2022-08-02T06:10:34.478718Z","iopub.status.idle":"2022-08-02T06:10:34.495685Z","shell.execute_reply.started":"2022-08-02T06:10:34.478682Z","shell.execute_reply":"2022-08-02T06:10:34.494548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\n\nsns.displot(\n    data=train.isna().melt(value_name=\"missing\"),\n    y=\"variable\",\n    hue=\"missing\",\n    multiple=\"fill\",\n    aspect=1.25\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:34.497603Z","iopub.execute_input":"2022-08-02T06:10:34.498069Z","iopub.status.idle":"2022-08-02T06:10:35.211008Z","shell.execute_reply.started":"2022-08-02T06:10:34.498023Z","shell.execute_reply":"2022-08-02T06:10:35.209699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.heatmap(train.isna().transpose(),\n            cmap=\"YlGnBu\",\n            cbar_kws={'label': 'Missing Data'})","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:35.212582Z","iopub.execute_input":"2022-08-02T06:10:35.213251Z","iopub.status.idle":"2022-08-02T06:10:36.315110Z","shell.execute_reply.started":"2022-08-02T06:10:35.213203Z","shell.execute_reply":"2022-08-02T06:10:36.313522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train.drop(['Transported','PassengerId','Name','Cabin','Destination'], axis=1)\ny = train['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:36.316924Z","iopub.execute_input":"2022-08-02T06:10:36.317837Z","iopub.status.idle":"2022-08-02T06:10:36.329982Z","shell.execute_reply.started":"2022-08-02T06:10:36.317784Z","shell.execute_reply":"2022-08-02T06:10:36.326881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = ['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\ncat_cols = ['HomePlanet', 'CryoSleep', 'VIP']\n\nfor i in num_cols:\n    X[i].fillna(np.mean(X[i]), inplace=True)\n    \nfor j in cat_cols:\n    X[j].fillna(X[i].mode()[0], inplace=True)\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:36.331774Z","iopub.execute_input":"2022-08-02T06:10:36.333108Z","iopub.status.idle":"2022-08-02T06:10:36.355925Z","shell.execute_reply.started":"2022-08-02T06:10:36.332947Z","shell.execute_reply":"2022-08-02T06:10:36.354490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\n\nsns.displot(\n    data=X.isna().melt(value_name=\"missing\"),\n    y=\"variable\",\n    hue=\"missing\",\n    multiple=\"fill\",\n    aspect=1.25\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:36.357430Z","iopub.execute_input":"2022-08-02T06:10:36.358504Z","iopub.status.idle":"2022-08-02T06:10:36.907599Z","shell.execute_reply.started":"2022-08-02T06:10:36.358456Z","shell.execute_reply":"2022-08-02T06:10:36.906479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = preprocessing.LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:36.912521Z","iopub.execute_input":"2022-08-02T06:10:36.912920Z","iopub.status.idle":"2022-08-02T06:10:36.919150Z","shell.execute_reply.started":"2022-08-02T06:10:36.912886Z","shell.execute_reply":"2022-08-02T06:10:36.917671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in num_cols:\n    X[i] = pd.qcut(X[i], q=5, duplicates='drop')\n    X[i] = le.fit_transform(X[i])\n\nX = pd.get_dummies(X, columns=['HomePlanet', 'CryoSleep', 'VIP'], drop_first=True)\n\nX.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:36.920924Z","iopub.execute_input":"2022-08-02T06:10:36.921844Z","iopub.status.idle":"2022-08-02T06:10:37.206613Z","shell.execute_reply.started":"2022-08-02T06:10:36.921797Z","shell.execute_reply":"2022-08-02T06:10:37.205468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:37.208250Z","iopub.execute_input":"2022-08-02T06:10:37.208672Z","iopub.status.idle":"2022-08-02T06:10:37.217929Z","shell.execute_reply.started":"2022-08-02T06:10:37.208637Z","shell.execute_reply":"2022-08-02T06:10:37.216929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SVC","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\n\nsvc = SVC(kernel='rbf')\nsvc.fit(X_train, y_train)\n\npredictions_svc = svc.predict(X_test)\nprint(f'Accuracy for SVC = {accuracy_score(y_test, predictions_svc)}')\nprint(f'\\nClassification report for SVC: \\n{classification_report(y_test, predictions_svc)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:37.219374Z","iopub.execute_input":"2022-08-02T06:10:37.220094Z","iopub.status.idle":"2022-08-02T06:10:40.064162Z","shell.execute_reply.started":"2022-08-02T06:10:37.220046Z","shell.execute_reply":"2022-08-02T06:10:40.062522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RandomForest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrfc = RandomForestClassifier(criterion='entropy', max_depth=8, n_estimators=400)\nrfc.fit(X_train, y_train)\n\npredictions_rfc = rfc.predict(X_test)\nprint(f'Accuracy for RandomForest = {accuracy_score(y_test, predictions_rfc)}')\nprint(f'\\nClassification report for RandomForest: \\n{classification_report(y_test, predictions_rfc)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:40.067145Z","iopub.execute_input":"2022-08-02T06:10:40.067648Z","iopub.status.idle":"2022-08-02T06:10:41.855617Z","shell.execute_reply.started":"2022-08-02T06:10:40.067599Z","shell.execute_reply":"2022-08-02T06:10:41.854473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Adaboost","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import AdaBoostClassifier\n\nada = AdaBoostClassifier(n_estimators=100, random_state=101)\nada.fit(X_train, y_train)\n\npredictions_ada = ada.predict(X_test)\nprint(f'Accuracy for ADA = {accuracy_score(y_test, predictions_ada)}')\nprint(f'\\nClassification report for ADA: \\n{classification_report(y_test, predictions_ada)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:14:28.392894Z","iopub.execute_input":"2022-08-02T06:14:28.393474Z","iopub.status.idle":"2022-08-02T06:14:28.887803Z","shell.execute_reply.started":"2022-08-02T06:14:28.393423Z","shell.execute_reply":"2022-08-02T06:14:28.886593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KNN","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nknn = KNeighborsClassifier(n_neighbors=1)\nknn.fit(X_train, y_train)\n\npredictions_knn = knn.predict(X_test)\nprint(f'Accuracy for KNN = {accuracy_score(y_test, predictions_knn)}')\nprint(f'\\nClassification report for KNN: \\n{classification_report(y_test, predictions_knn)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:41.856950Z","iopub.execute_input":"2022-08-02T06:10:41.857282Z","iopub.status.idle":"2022-08-02T06:10:41.968805Z","shell.execute_reply.started":"2022-08-02T06:10:41.857251Z","shell.execute_reply":"2022-08-02T06:10:41.967412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gaussian NB","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\ngnb = GaussianNB()\ngnb.fit(X_train, y_train)\n\npredictions_gnb = gnb.predict(X_test)\nprint(f'Accuracy for GaussianNB = {accuracy_score(y_test, predictions_gnb)}')\nprint(f'\\nClassification report for GaussianNB: \\n{classification_report(y_test, predictions_gnb)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:41.970464Z","iopub.execute_input":"2022-08-02T06:10:41.971568Z","iopub.status.idle":"2022-08-02T06:10:41.998553Z","shell.execute_reply.started":"2022-08-02T06:10:41.971530Z","shell.execute_reply":"2022-08-02T06:10:41.997304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGB","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\nxgb = XGBClassifier()\nxgb.fit(X_train, y_train)\n\npredictions_xgb = xgb.predict(X_test)\nprint(f'Accuracy for XGB = {accuracy_score(y_test, predictions_xgb)}')\nprint(f'\\nClassification report for XGB: \\n{classification_report(y_test, predictions_xgb)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:42.000088Z","iopub.execute_input":"2022-08-02T06:10:42.000528Z","iopub.status.idle":"2022-08-02T06:10:42.744015Z","shell.execute_reply.started":"2022-08-02T06:10:42.000494Z","shell.execute_reply":"2022-08-02T06:10:42.742731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nclf = LogisticRegression(random_state=0)\nclf.fit(X_train, y_train)\n\npredictions_clf = clf.predict(X_test)\nprint(f'Accuracy for CLF = {accuracy_score(y_test, predictions_clf)}')\nprint(f'\\nClassification report for CLF: \\n{classification_report(y_test, predictions_clf)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:42.745574Z","iopub.execute_input":"2022-08-02T06:10:42.746707Z","iopub.status.idle":"2022-08-02T06:10:42.814833Z","shell.execute_reply.started":"2022-08-02T06:10:42.746666Z","shell.execute_reply":"2022-08-02T06:10:42.813852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Perceptron","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Perceptron\n\nper = Perceptron(tol=1e-3, random_state=0)\nper.fit(X_train, y_train)\n\npredictions_per = per.predict(X_test)\nprint(f'Accuracy for Perceptron model = {accuracy_score(y_test, predictions_per)}')\nprint(f'\\nClassification report for Perceptron model: \\n{classification_report(y_test, predictions_per)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:42.816540Z","iopub.execute_input":"2022-08-02T06:10:42.817186Z","iopub.status.idle":"2022-08-02T06:10:42.853161Z","shell.execute_reply.started":"2022-08-02T06:10:42.817138Z","shell.execute_reply":"2022-08-02T06:10:42.851897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple Dense Neural Network","metadata":{}},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras import layers\n\nmodel = keras.Sequential([\n    layers.Dense(units=256, activation='relu', input_shape=[11]),\n    layers.Dense(units=256, activation='relu'),\n    layers.Dense(units=1, activation='sigmoid'),\n])\n\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy'],\n)\n\nearly_stopping = keras.callbacks.EarlyStopping(\n    patience=10,\n    min_delta=0.001,\n    restore_best_weights=True,\n)\n\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_test, y_test),\n    batch_size=512,\n    epochs=1000,\n    callbacks=[early_stopping],\n    verbose=0,\n)\n\nhistory_df = pd.DataFrame(history.history)\n\nprint((\"Best Validation Accuracy: {:0.4f}\")\\\n      .format(history_df['val_binary_accuracy'].max()))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:42.854912Z","iopub.execute_input":"2022-08-02T06:10:42.855613Z","iopub.status.idle":"2022-08-02T06:10:53.142552Z","shell.execute_reply.started":"2022-08-02T06:10:42.855566Z","shell.execute_reply":"2022-08-02T06:10:53.141229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NN with Dropout and BatchNormalization","metadata":{}},{"cell_type":"code","source":"model = keras.Sequential([\n    layers.Dense(units=1024, activation='relu', input_shape=[11]),\n    layers.Dropout(0.3),\n    layers.BatchNormalization(),\n    layers.Dense(units=1024, activation='relu'),\n    layers.Dropout(0.3),\n    layers.BatchNormalization(),\n    layers.Dense(units=1024, activation='relu'),\n    layers.Dropout(0.3),\n    layers.BatchNormalization(),\n    layers.Dense(units=1, activation='sigmoid'),\n])\n\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy'],\n)\n\nearly_stopping = keras.callbacks.EarlyStopping(\n    patience=10,\n    min_delta=0.001,\n    restore_best_weights=True,\n)\n\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_test, y_test),\n    batch_size=512,\n    epochs=1000,\n    callbacks=[early_stopping],\n    verbose=0,\n)\n\nhistory_df = pd.DataFrame(history.history)\n\nprint((\"Best Validation Accuracy: {:0.4f}\")\\\n      .format(history_df['val_binary_accuracy'].max()))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:10:53.144099Z","iopub.execute_input":"2022-08-02T06:10:53.144486Z","iopub.status.idle":"2022-08-02T06:12:06.661902Z","shell.execute_reply.started":"2022-08-02T06:10:53.144450Z","shell.execute_reply":"2022-08-02T06:12:06.660662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}