{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T07:10:14.870011Z","iopub.execute_input":"2022-08-01T07:10:14.870547Z","iopub.status.idle":"2022-08-01T07:10:14.885083Z","shell.execute_reply.started":"2022-08-01T07:10:14.870442Z","shell.execute_reply":"2022-08-01T07:10:14.883421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Introduction to dataset","metadata":{}},{"cell_type":"markdown","source":"****Project goal:****\n> We have to predict whether the passanger was transported or not","metadata":{}},{"cell_type":"markdown","source":"****Datasets****:\n> We already have split dataset into train and test set\n* Train.csv\n* Test.csv","metadata":{}},{"cell_type":"markdown","source":"**Machine learning models we'll be using here**:\n\n1. Logistic regression\n2. KNearestNeighbors\n3. Decision tree\n4. Random Forest\n5. Linear Support Vector Machine (SVM)","metadata":{}},{"cell_type":"markdown","source":"# 2. Importing modules","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:14.969502Z","iopub.execute_input":"2022-08-01T07:10:14.969904Z","iopub.status.idle":"2022-08-01T07:10:14.977483Z","shell.execute_reply.started":"2022-08-01T07:10:14.969871Z","shell.execute_reply":"2022-08-01T07:10:14.976142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Reading dataset","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('..//input//spaceship-titanic//train.csv')\ntest_df = pd.read_csv('..//input//spaceship-titanic//test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:14.985322Z","iopub.execute_input":"2022-08-01T07:10:14.987171Z","iopub.status.idle":"2022-08-01T07:10:15.056090Z","shell.execute_reply.started":"2022-08-01T07:10:14.987098Z","shell.execute_reply":"2022-08-01T07:10:15.054625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.058559Z","iopub.execute_input":"2022-08-01T07:10:15.058883Z","iopub.status.idle":"2022-08-01T07:10:15.091720Z","shell.execute_reply.started":"2022-08-01T07:10:15.058852Z","shell.execute_reply":"2022-08-01T07:10:15.090191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.093361Z","iopub.execute_input":"2022-08-01T07:10:15.093687Z","iopub.status.idle":"2022-08-01T07:10:15.130319Z","shell.execute_reply.started":"2022-08-01T07:10:15.093656Z","shell.execute_reply":"2022-08-01T07:10:15.128794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.132497Z","iopub.execute_input":"2022-08-01T07:10:15.133364Z","iopub.status.idle":"2022-08-01T07:10:15.163598Z","shell.execute_reply.started":"2022-08-01T07:10:15.133314Z","shell.execute_reply":"2022-08-01T07:10:15.161984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> There are some missing values, so we have to deal with that first","metadata":{}},{"cell_type":"markdown","source":"# 4. Dealing with missing values","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.168977Z","iopub.execute_input":"2022-08-01T07:10:15.169382Z","iopub.status.idle":"2022-08-01T07:10:15.175862Z","shell.execute_reply.started":"2022-08-01T07:10:15.169346Z","shell.execute_reply":"2022-08-01T07:10:15.174710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = SimpleImputer(strategy='median')\n\ntrain_df[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']] = imputer.fit_transform(train_df[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']])\ntest_df[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']] = imputer.fit_transform(test_df[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']])\n\nimputer = SimpleImputer(strategy='most_frequent')\n\ntrain_df[['HomePlanet', 'CryoSleep', 'Cabin', 'Destination', 'VIP']] = imputer.fit_transform(train_df[['HomePlanet', 'CryoSleep', 'Cabin', 'Destination', 'VIP']])\ntest_df[['HomePlanet', 'CryoSleep', 'Cabin', 'Destination', 'VIP']] = imputer.fit_transform(test_df[['HomePlanet', 'CryoSleep', 'Cabin', 'Destination', 'VIP']])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.177795Z","iopub.execute_input":"2022-08-01T07:10:15.178525Z","iopub.status.idle":"2022-08-01T07:10:15.249612Z","shell.execute_reply.started":"2022-08-01T07:10:15.178481Z","shell.execute_reply":"2022-08-01T07:10:15.248338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.251571Z","iopub.execute_input":"2022-08-01T07:10:15.252344Z","iopub.status.idle":"2022-08-01T07:10:15.281509Z","shell.execute_reply.started":"2022-08-01T07:10:15.252293Z","shell.execute_reply":"2022-08-01T07:10:15.279955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Visualizing data","metadata":{}},{"cell_type":"code","source":"# sns.set_theme(style='darkgrid')\n# mask = ((train_df.dtypes == 'float64') | (train_df.dtypes == 'bool'))\n# num_columns = train_df.loc[:, mask]\n# sns.pairplot(data=num_columns, hue='Transported')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.283732Z","iopub.execute_input":"2022-08-01T07:10:15.284203Z","iopub.status.idle":"2022-08-01T07:10:15.290609Z","shell.execute_reply.started":"2022-08-01T07:10:15.284158Z","shell.execute_reply":"2022-08-01T07:10:15.289062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style=\"darkgrid\")\nsns.countplot(data=train_df, x='HomePlanet', hue='Transported')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.292683Z","iopub.execute_input":"2022-08-01T07:10:15.293425Z","iopub.status.idle":"2022-08-01T07:10:15.594262Z","shell.execute_reply.started":"2022-08-01T07:10:15.293379Z","shell.execute_reply":"2022-08-01T07:10:15.592483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style=\"darkgrid\")\nsns.countplot(data=train_df, x='CryoSleep', hue='Transported')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.596463Z","iopub.execute_input":"2022-08-01T07:10:15.597066Z","iopub.status.idle":"2022-08-01T07:10:15.847760Z","shell.execute_reply.started":"2022-08-01T07:10:15.596990Z","shell.execute_reply":"2022-08-01T07:10:15.846677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style=\"darkgrid\")\nsns.countplot(data=train_df, x='Destination', hue='Transported')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:15.849203Z","iopub.execute_input":"2022-08-01T07:10:15.849558Z","iopub.status.idle":"2022-08-01T07:10:16.131370Z","shell.execute_reply.started":"2022-08-01T07:10:15.849525Z","shell.execute_reply":"2022-08-01T07:10:16.129891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style=\"darkgrid\")\nsns.countplot(data=train_df, x='VIP', hue='Transported')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:16.133964Z","iopub.execute_input":"2022-08-01T07:10:16.134628Z","iopub.status.idle":"2022-08-01T07:10:16.365306Z","shell.execute_reply.started":"2022-08-01T07:10:16.134577Z","shell.execute_reply":"2022-08-01T07:10:16.364185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Manipulating and cleaning and more visualizing data","metadata":{}},{"cell_type":"code","source":"train_df[['Deck', 'Num', 'Side']] = train_df['Cabin'].str.split('/', expand=True)\ntrain_df.drop('Cabin', axis=1, inplace=True)\n\ntest_df[['Deck', 'Num', 'Side']] = test_df['Cabin'].str.split('/', expand=True)\ntest_df.drop('Cabin', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:16.366861Z","iopub.execute_input":"2022-08-01T07:10:16.367236Z","iopub.status.idle":"2022-08-01T07:10:16.415784Z","shell.execute_reply.started":"2022-08-01T07:10:16.367201Z","shell.execute_reply":"2022-08-01T07:10:16.414632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop('Name', axis=1, inplace=True)\ntest_df.drop('Name', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:16.422827Z","iopub.execute_input":"2022-08-01T07:10:16.423502Z","iopub.status.idle":"2022-08-01T07:10:16.432803Z","shell.execute_reply.started":"2022-08-01T07:10:16.423463Z","shell.execute_reply":"2022-08-01T07:10:16.431351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['HomePlanet', 'Destination', 'Side', 'Deck']] = train_df[['HomePlanet', 'Destination', 'Side', 'Deck']].astype('category')\ntrain_df['Num'] = train_df['Num'].astype('int64')\ntrain_df[['CryoSleep', 'VIP']] = train_df[['CryoSleep', 'VIP']].astype('bool')\n\ntest_df[['HomePlanet', 'Destination', 'Side', 'Deck']] = test_df[['HomePlanet', 'Destination', 'Side', 'Deck']].astype('category')\ntest_df['Num'] = test_df['Num'].astype('int64')\ntest_df[['CryoSleep', 'VIP']] = test_df[['CryoSleep', 'VIP']].astype('bool')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:16.434325Z","iopub.execute_input":"2022-08-01T07:10:16.434906Z","iopub.status.idle":"2022-08-01T07:10:16.478471Z","shell.execute_reply.started":"2022-08-01T07:10:16.434871Z","shell.execute_reply":"2022-08-01T07:10:16.477437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:16.480184Z","iopub.execute_input":"2022-08-01T07:10:16.480812Z","iopub.status.idle":"2022-08-01T07:10:16.501974Z","shell.execute_reply.started":"2022-08-01T07:10:16.480777Z","shell.execute_reply":"2022-08-01T07:10:16.500514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:16.503707Z","iopub.execute_input":"2022-08-01T07:10:16.504099Z","iopub.status.idle":"2022-08-01T07:10:16.523612Z","shell.execute_reply.started":"2022-08-01T07:10:16.504059Z","shell.execute_reply":"2022-08-01T07:10:16.522787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style=\"darkgrid\")\nsns.countplot(data=train_df, x='Side', hue='Transported')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:16.525180Z","iopub.execute_input":"2022-08-01T07:10:16.525821Z","iopub.status.idle":"2022-08-01T07:10:16.734599Z","shell.execute_reply.started":"2022-08-01T07:10:16.525787Z","shell.execute_reply":"2022-08-01T07:10:16.733684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style=\"darkgrid\")\nsns.countplot(data=train_df, x='Deck', hue='Transported')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:16.736194Z","iopub.execute_input":"2022-08-01T07:10:16.736830Z","iopub.status.idle":"2022-08-01T07:10:17.035679Z","shell.execute_reply.started":"2022-08-01T07:10:16.736794Z","shell.execute_reply":"2022-08-01T07:10:17.034362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. Transforming categories and bool columns into int columns","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder, OneHotEncoder, StandardScaler","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.037618Z","iopub.execute_input":"2022-08-01T07:10:17.038081Z","iopub.status.idle":"2022-08-01T07:10:17.044332Z","shell.execute_reply.started":"2022-08-01T07:10:17.038016Z","shell.execute_reply":"2022-08-01T07:10:17.043027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.046384Z","iopub.execute_input":"2022-08-01T07:10:17.046814Z","iopub.status.idle":"2022-08-01T07:10:17.069838Z","shell.execute_reply.started":"2022-08-01T07:10:17.046762Z","shell.execute_reply":"2022-08-01T07:10:17.068354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = ['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\nscaler = StandardScaler()\nscaler.fit(train_df[mask])\ntrain_df[mask] = scaler.transform(train_df[mask])\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.071645Z","iopub.execute_input":"2022-08-01T07:10:17.072005Z","iopub.status.idle":"2022-08-01T07:10:17.107308Z","shell.execute_reply.started":"2022-08-01T07:10:17.071974Z","shell.execute_reply":"2022-08-01T07:10:17.106090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_encoder = LabelEncoder()\n\nfor x in list(train_df.columns):\n    if train_df[x].dtype=='category':\n        train_df[x]=label_encoder.fit_transform(train_df[x])\n\nfor x in list(test_df.columns):\n    if test_df[x].dtype=='category':\n        test_df[x]=label_encoder.fit_transform(test_df[x])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.108544Z","iopub.execute_input":"2022-08-01T07:10:17.108950Z","iopub.status.idle":"2022-08-01T07:10:17.148581Z","shell.execute_reply.started":"2022-08-01T07:10:17.108913Z","shell.execute_reply":"2022-08-01T07:10:17.147155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"onehot = OneHotEncoder()\n\nfor x in list(train_df.columns):\n    if train_df[x].dtype=='bool':\n        train_df[x]=label_encoder.fit_transform(train_df[x])\n        \nfor x in list(test_df.columns):\n    if test_df[x].dtype=='bool':\n        test_df[x]=label_encoder.fit_transform(test_df[x])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.150705Z","iopub.execute_input":"2022-08-01T07:10:17.151209Z","iopub.status.idle":"2022-08-01T07:10:17.166981Z","shell.execute_reply.started":"2022-08-01T07:10:17.151165Z","shell.execute_reply":"2022-08-01T07:10:17.165382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.169813Z","iopub.execute_input":"2022-08-01T07:10:17.170586Z","iopub.status.idle":"2022-08-01T07:10:17.194612Z","shell.execute_reply.started":"2022-08-01T07:10:17.170526Z","shell.execute_reply":"2022-08-01T07:10:17.193437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.196444Z","iopub.execute_input":"2022-08-01T07:10:17.197111Z","iopub.status.idle":"2022-08-01T07:10:17.216760Z","shell.execute_reply.started":"2022-08-01T07:10:17.197070Z","shell.execute_reply":"2022-08-01T07:10:17.215169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 8. Models selection","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nimport xgboost as xgb\n\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.model_selection import train_test_split, GridSearchCV, KFold","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:11:49.808179Z","iopub.execute_input":"2022-08-01T07:11:49.808633Z","iopub.status.idle":"2022-08-01T07:11:49.942357Z","shell.execute_reply.started":"2022-08-01T07:11:49.808597Z","shell.execute_reply":"2022-08-01T07:11:49.940873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(['PassengerId', 'Transported'], axis=1)\ny = train_df['Transported']\nX_valid = test_df.drop('PassengerId', axis=1)\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=101)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.228340Z","iopub.execute_input":"2022-08-01T07:10:17.229357Z","iopub.status.idle":"2022-08-01T07:10:17.613970Z","shell.execute_reply.started":"2022-08-01T07:10:17.229315Z","shell.execute_reply":"2022-08-01T07:10:17.612789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**8.1 LogisticRegression**","metadata":{}},{"cell_type":"code","source":"logreg = LogisticRegression(max_iter=500)\nlogreg.fit(X_train, y_train)\ny_pred = logreg.predict(X_test)\n\nprint(classification_report(y_test, y_pred))\nprint(confusion_matrix(y_test, y_pred))\n\nacc = accuracy_score(y_pred, y_test)\nprint('Logistic Regression accuracy is: {:.2f}%'.format(acc*100))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:17.615767Z","iopub.execute_input":"2022-08-01T07:10:17.616147Z","iopub.status.idle":"2022-08-01T07:10:18.153792Z","shell.execute_reply.started":"2022-08-01T07:10:17.616106Z","shell.execute_reply":"2022-08-01T07:10:18.129151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****8.2 NearestNeighbors****","metadata":{}},{"cell_type":"code","source":"scoreList = []\nfor i in range(1,10):\n    NN = KNeighborsClassifier(n_neighbors = i)\n    NN.fit(X_train, y_train)\n    scoreList.append(NN.score(X_test, y_test))\n    \nplt.plot(range(1,10), scoreList)\nplt.xlabel(\"K value\")\nplt.ylabel(\"Score\")\nplt.show()\nacc = max(scoreList)\n\nNN = KNeighborsClassifier(n_neighbors = 1)\nNN.fit(X_train, y_train)\nprint(classification_report(y_test, NN.predict(X_test)))\nprint(confusion_matrix(y_test, NN.predict(X_test)))\n\nprint(\"KNN max accuracy is: {:.2f}%\".format(acc*100))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:18.155543Z","iopub.execute_input":"2022-08-01T07:10:18.156110Z","iopub.status.idle":"2022-08-01T07:10:19.862507Z","shell.execute_reply.started":"2022-08-01T07:10:18.156061Z","shell.execute_reply":"2022-08-01T07:10:19.860741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**8.3 DecisionTreeClassifier**","metadata":{}},{"cell_type":"code","source":"params = {'min_samples_split': [2, 4, 6, 8, 10],\n          'max_features': np.linspace(1, 13, 13, dtype=int),\n          'max_leaf_nodes': np.linspace(10, 100, 10, dtype=int)}\nkf = KFold(n_splits=2, shuffle=True, random_state=101)\ndt = DecisionTreeClassifier()\n\ndtcv = GridSearchCV(estimator=dt, \n                    param_grid=params, \n                    cv=kf)\n\ndtcv.fit(X_train, y_train)\n\nprint('Tuned Decision Tree best score {}'.format(dtcv.best_score_))\nprint('Tuned Decision Tree best params {}'.format(dtcv.best_estimator_))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:19.864578Z","iopub.execute_input":"2022-08-01T07:10:19.865141Z","iopub.status.idle":"2022-08-01T07:10:30.378107Z","shell.execute_reply.started":"2022-08-01T07:10:19.865094Z","shell.execute_reply":"2022-08-01T07:10:30.375721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt = DecisionTreeClassifier(max_features=11, max_leaf_nodes=30, min_samples_split=10)\ndt.fit(X_train, y_train)\n\ny_pred = dt.predict(X_test)\n\nprint(classification_report(y_test, y_pred))\nprint(confusion_matrix(y_test, y_pred))\n\n\nacc = accuracy_score(y_pred,y_test)\nprint('Decision Tree accuracy is: {:.2f}%'.format(acc*100))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:30.379652Z","iopub.status.idle":"2022-08-01T07:10:30.380938Z","shell.execute_reply.started":"2022-08-01T07:10:30.380584Z","shell.execute_reply":"2022-08-01T07:10:30.380620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**8.4 RandomForestClassifier**","metadata":{}},{"cell_type":"code","source":"params = {'max_depth': np.linspace(1, 14, 3, dtype=int),\n          'n_estimators': [100],\n          'max_features': np.linspace(1, 13, 13, dtype=int),\n          'max_leaf_nodes': np.linspace(10, 100, 10, dtype=int)}\nkf = KFold(n_splits=2, shuffle=True, random_state=101)\nrf = RandomForestClassifier()\n\nrfcv = GridSearchCV(estimator=rf, \n                    param_grid=params, \n                    cv=kf)\n\nrfcv.fit(X_train, y_train)\n\nprint('Tuned Random Forest best score {}'.format(rfcv.best_score_))\nprint('Tuned Random Forest best params {}'.format(rfcv.best_estimator_))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:30.382883Z","iopub.status.idle":"2022-08-01T07:10:30.383574Z","shell.execute_reply.started":"2022-08-01T07:10:30.383254Z","shell.execute_reply":"2022-08-01T07:10:30.383283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = RandomForestClassifier(max_depth=14, max_features=3, max_leaf_nodes=90)\nrf.fit(X_train, y_train)\n\ny_pred = rf.predict(X_test)\n\nprint(classification_report(y_test, y_pred))\nprint(confusion_matrix(y_test, y_pred))\n\nacc = accuracy_score(y_pred,y_test)\nprint('Random Forest accuracy is: {:.2f}%'.format(acc*100))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:30.386698Z","iopub.status.idle":"2022-08-01T07:10:30.387432Z","shell.execute_reply.started":"2022-08-01T07:10:30.387094Z","shell.execute_reply":"2022-08-01T07:10:30.387128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**8.5 SVC**","metadata":{}},{"cell_type":"code","source":"params = {'C': np.linspace(1, 10, 1, dtype=int)}\nkf = KFold(n_splits=5, shuffle=True, random_state=101)\nsvc = SVC()\n\nsvc_cv= GridSearchCV(estimator=svc, \n                    param_grid=params, \n                    cv=kf)\n\nsvc_cv.fit(X_train, y_train)\n\nprint('Tuned SVC best score {}'.format(svc_cv.best_score_))\nprint('Tuned SVC best params {}'.format(svc_cv.best_estimator_))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:30.389320Z","iopub.status.idle":"2022-08-01T07:10:30.390264Z","shell.execute_reply.started":"2022-08-01T07:10:30.389970Z","shell.execute_reply":"2022-08-01T07:10:30.389995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc = SVC()\nsvc.fit(X_train, y_train)\n\ny_pred = svc.predict(X_test)\n\nprint(classification_report(y_test, y_pred))\nprint(confusion_matrix(y_test, y_pred))\n\nacc = accuracy_score(y_pred,y_test)\nprint('SVC accuracy is: {:.2f}%'.format(acc*100))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:30.391759Z","iopub.status.idle":"2022-08-01T07:10:30.392663Z","shell.execute_reply.started":"2022-08-01T07:10:30.392424Z","shell.execute_reply":"2022-08-01T07:10:30.392448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**8.5 XGBooster**","metadata":{}},{"cell_type":"code","source":"model = xgb.XGBClassifier(objective='binary:logistic', \n                  n_estimators=100, \n                  max_depth=3,\n                  eta=0.8,\n                  gamma=2\n                         )\n\nmodel.fit(X_train, y_train)\n\ny_pred = model.predict(X_test)\n\nacc = accuracy_score(y_pred,y_test)\nprint('SVC accuracy is: {:.2f}%'.format(acc*100))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:19:27.538139Z","iopub.execute_input":"2022-08-01T07:19:27.538592Z","iopub.status.idle":"2022-08-01T07:19:29.129877Z","shell.execute_reply.started":"2022-08-01T07:19:27.538558Z","shell.execute_reply":"2022-08-01T07:19:29.121995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 9. Output","metadata":{}},{"cell_type":"code","source":"model = xgb.XGBClassifier(objective='binary:logistic', \n                  n_estimators=100, \n                  max_depth=3,\n                  eta=0.8,\n                  gamma=2\n                         )\nmodel.fit(X_train, y_train)\ny_pred = model.predict(test_df.drop('PassengerId', axis=1))\nps_id = test_df['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:30.394173Z","iopub.status.idle":"2022-08-01T07:10:30.394632Z","shell.execute_reply.started":"2022-08-01T07:10:30.394421Z","shell.execute_reply":"2022-08-01T07:10:30.394442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId': ps_id,\n                       'Transported': y_pred.astype('bool')})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T07:10:30.396483Z","iopub.status.idle":"2022-08-01T07:10:30.396945Z","shell.execute_reply.started":"2022-08-01T07:10:30.396729Z","shell.execute_reply":"2022-08-01T07:10:30.396749Z"},"trusted":true},"execution_count":null,"outputs":[]}]}