{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-14T16:54:32.855078Z","iopub.execute_input":"2022-08-14T16:54:32.855510Z","iopub.status.idle":"2022-08-14T16:54:32.865908Z","shell.execute_reply.started":"2022-08-14T16:54:32.855473Z","shell.execute_reply":"2022-08-14T16:54:32.864244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Loading the data</h3>","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv', index_col='PassengerId')\ntest_df = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv', index_col='PassengerId')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:32.891935Z","iopub.execute_input":"2022-08-14T16:54:32.892790Z","iopub.status.idle":"2022-08-14T16:54:32.979303Z","shell.execute_reply.started":"2022-08-14T16:54:32.892746Z","shell.execute_reply":"2022-08-14T16:54:32.978192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Checking basic info</h3>","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:32.981662Z","iopub.execute_input":"2022-08-14T16:54:32.982266Z","iopub.status.idle":"2022-08-14T16:54:33.011721Z","shell.execute_reply.started":"2022-08-14T16:54:32.982218Z","shell.execute_reply":"2022-08-14T16:54:33.010898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:33.013104Z","iopub.execute_input":"2022-08-14T16:54:33.013652Z","iopub.status.idle":"2022-08-14T16:54:33.036344Z","shell.execute_reply.started":"2022-08-14T16:54:33.013620Z","shell.execute_reply":"2022-08-14T16:54:33.035001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pandas_profiling import ProfileReport\n\nprofile = ProfileReport(train_df, title=\"Profiling Report\")\nprofile","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:33.039090Z","iopub.execute_input":"2022-08-14T16:54:33.039628Z","iopub.status.idle":"2022-08-14T16:54:54.088957Z","shell.execute_reply.started":"2022-08-14T16:54:33.039578Z","shell.execute_reply":"2022-08-14T16:54:54.088019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Some feature engineering</h3>","metadata":{}},{"cell_type":"code","source":"train_df.drop('Name', axis=1, inplace=True)\ntest_df.drop('Name', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.090735Z","iopub.execute_input":"2022-08-14T16:54:54.091261Z","iopub.status.idle":"2022-08-14T16:54:54.100214Z","shell.execute_reply.started":"2022-08-14T16:54:54.091228Z","shell.execute_reply":"2022-08-14T16:54:54.099240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Transported'].replace(False, 0, inplace=True)\ntrain_df['Transported'].replace(True, 1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.101657Z","iopub.execute_input":"2022-08-14T16:54:54.102274Z","iopub.status.idle":"2022-08-14T16:54:54.133889Z","shell.execute_reply.started":"2022-08-14T16:54:54.102240Z","shell.execute_reply":"2022-08-14T16:54:54.132498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Let's separate the cabin columns in three new features</h3>","metadata":{}},{"cell_type":"code","source":"train_df[['deck','num', 'side']] = train_df['Cabin'].str.split('/', expand=True)\ntest_df[['deck','num', 'side']] = test_df['Cabin'].str.split('/', expand=True)\n\ntrain_df.drop('Cabin', axis=1, inplace=True)\ntest_df.drop('Cabin', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.136324Z","iopub.execute_input":"2022-08-14T16:54:54.136965Z","iopub.status.idle":"2022-08-14T16:54:54.175240Z","shell.execute_reply.started":"2022-08-14T16:54:54.136926Z","shell.execute_reply":"2022-08-14T16:54:54.174385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"object_cols = [col for col in train_df.columns if train_df[col].dtype == 'object' or train_df[col].dtype == 'category']\nnumeric_cols = [col for col in train_df.columns if train_df[col].dtype == 'float64']\n\nprint(f'Object cols -- {object_cols}')\nprint(f'Numeric cols -- {numeric_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.176465Z","iopub.execute_input":"2022-08-14T16:54:54.176978Z","iopub.status.idle":"2022-08-14T16:54:54.184206Z","shell.execute_reply.started":"2022-08-14T16:54:54.176946Z","shell.execute_reply":"2022-08-14T16:54:54.183377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Sum of spent value by passenger, creating a new feature</h3>","metadata":{}},{"cell_type":"code","source":"col_to_sum = ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\n\ntrain_df['SumSpends'] = train_df[col_to_sum].sum(axis=1)\ntest_df['SumSpends'] = test_df[col_to_sum].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.185577Z","iopub.execute_input":"2022-08-14T16:54:54.186150Z","iopub.status.idle":"2022-08-14T16:54:54.201669Z","shell.execute_reply.started":"2022-08-14T16:54:54.186102Z","shell.execute_reply":"2022-08-14T16:54:54.200773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Checking null and object columns</h3>","metadata":{}},{"cell_type":"code","source":"null_cols = train_df.isnull().sum().sort_values(ascending=False)\nnull_cols = list(null_cols[null_cols>1].index)\nnull_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.203007Z","iopub.execute_input":"2022-08-14T16:54:54.204099Z","iopub.status.idle":"2022-08-14T16:54:54.217861Z","shell.execute_reply.started":"2022-08-14T16:54:54.204064Z","shell.execute_reply":"2022-08-14T16:54:54.216740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[object_cols] = train_df[object_cols].astype('category')\ntest_df[object_cols] = test_df[object_cols].astype('category')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.219583Z","iopub.execute_input":"2022-08-14T16:54:54.220291Z","iopub.status.idle":"2022-08-14T16:54:54.255631Z","shell.execute_reply.started":"2022-08-14T16:54:54.220253Z","shell.execute_reply":"2022-08-14T16:54:54.254810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train DF shape: {train_df.shape}')\nprint(f'Test DF shape: {test_df.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.256824Z","iopub.execute_input":"2022-08-14T16:54:54.257599Z","iopub.status.idle":"2022-08-14T16:54:54.263202Z","shell.execute_reply.started":"2022-08-14T16:54:54.257564Z","shell.execute_reply":"2022-08-14T16:54:54.261946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Encoding the categorical variables</h3>","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\noc = OrdinalEncoder()\n\ndf_for_encode = pd.concat([train_df, test_df])\n\ndf_for_encode[object_cols] = df_for_encode[object_cols].astype('category')\n\ndf_for_encode[object_cols] = oc.fit_transform(df_for_encode[object_cols])\n\ndel train_df, test_df\n\ntrain_df = df_for_encode.iloc[:8693, :]\ntest_df = df_for_encode.iloc[8693: , :]\n\ndel df_for_encode\n\ntest_df.drop('Transported', inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.267960Z","iopub.execute_input":"2022-08-14T16:54:54.268829Z","iopub.status.idle":"2022-08-14T16:54:54.364065Z","shell.execute_reply.started":"2022-08-14T16:54:54.268792Z","shell.execute_reply":"2022-08-14T16:54:54.362817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train DF shape: {train_df.shape}')\nprint(f'Test DF shape: {test_df.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.365429Z","iopub.execute_input":"2022-08-14T16:54:54.365850Z","iopub.status.idle":"2022-08-14T16:54:54.372524Z","shell.execute_reply.started":"2022-08-14T16:54:54.365813Z","shell.execute_reply":"2022-08-14T16:54:54.371216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\n\n\nct = ColumnTransformer([(\"imp\", SimpleImputer(strategy='mean'), null_cols)])\n    \ntrain_df[null_cols] = ct.fit_transform(train_df[null_cols])\ntest_df[null_cols] = ct.fit_transform(test_df[null_cols])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.374288Z","iopub.execute_input":"2022-08-14T16:54:54.375769Z","iopub.status.idle":"2022-08-14T16:54:54.540210Z","shell.execute_reply.started":"2022-08-14T16:54:54.375719Z","shell.execute_reply":"2022-08-14T16:54:54.539006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Prearing the dataset for modeling</h3>","metadata":{}},{"cell_type":"code","source":"X = train_df.copy()\ny = X.pop('Transported')\n\nfrom sklearn.model_selection import cross_val_score, train_test_split, GridSearchCV\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=23)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T16:54:54.541741Z","iopub.execute_input":"2022-08-14T16:54:54.542132Z","iopub.status.idle":"2022-08-14T16:54:54.555427Z","shell.execute_reply.started":"2022-08-14T16:54:54.542096Z","shell.execute_reply":"2022-08-14T16:54:54.553958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Testing 4 models without hyperparameter tunning</h3>","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import accuracy_score\nfrom catboost import CatBoostClassifier\n\ndef predict_and_acc(model, verbose=None):\n    if verbose == None:\n        model = model()\n        model.fit(X_train, y_train)\n        predict = model.predict(X_test)\n        cvs = cross_val_score(model, X, y, cv=4)\n        print(f'The accuracy score of {str(model)} is {float(accuracy_score(y_test, predict))}')\n        print(f'The cross validation of {str(model)} is:{cvs} with mean of {cvs.mean()}')\n    else:\n        model = model(verbose=verbose)\n        model.fit(X_train, y_train)\n        predict = model.predict(X_test)\n        cvs = cross_val_score(model, X, y, cv=4)\n        print(f'The accuracy score of {str(model)} is {float(accuracy_score(y_test, predict))}')\n        print(f'The cross validation of {str(model)} is:{cvs} with mean of {cvs.mean()}')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:03:45.837268Z","iopub.execute_input":"2022-08-14T17:03:45.837726Z","iopub.status.idle":"2022-08-14T17:03:45.847222Z","shell.execute_reply.started":"2022-08-14T17:03:45.837665Z","shell.execute_reply":"2022-08-14T17:03:45.846074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_and_acc(RandomForestClassifier, None)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:02:51.227067Z","iopub.execute_input":"2022-08-14T17:02:51.228317Z","iopub.status.idle":"2022-08-14T17:02:56.403616Z","shell.execute_reply.started":"2022-08-14T17:02:51.228262Z","shell.execute_reply":"2022-08-14T17:02:56.402235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_and_acc(AdaBoostClassifier)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:03:04.517790Z","iopub.execute_input":"2022-08-14T17:03:04.518194Z","iopub.status.idle":"2022-08-14T17:03:06.492062Z","shell.execute_reply.started":"2022-08-14T17:03:04.518154Z","shell.execute_reply":"2022-08-14T17:03:06.490935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_and_acc(LGBMClassifier)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:04:08.874771Z","iopub.execute_input":"2022-08-14T17:04:08.877613Z","iopub.status.idle":"2022-08-14T17:04:09.842180Z","shell.execute_reply.started":"2022-08-14T17:04:08.877564Z","shell.execute_reply":"2022-08-14T17:04:09.840857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_and_acc(CatBoostClassifier, verbose=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:03:47.865648Z","iopub.execute_input":"2022-08-14T17:03:47.866081Z","iopub.status.idle":"2022-08-14T17:04:05.989599Z","shell.execute_reply.started":"2022-08-14T17:03:47.866046Z","shell.execute_reply":"2022-08-14T17:04:05.988381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Backward Feature Selection for the best model</h3>","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_selection import SequentialFeatureSelector\n\nmodel_fs = CatBoostClassifier(verbose=False)\nsf = SequentialFeatureSelector(model_fs, scoring='accuracy', direction = 'backward')\nsf.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:04:18.400014Z","iopub.execute_input":"2022-08-14T17:04:18.400446Z","iopub.status.idle":"2022-08-14T17:27:14.786798Z","shell.execute_reply.started":"2022-08-14T17:04:18.400410Z","shell.execute_reply":"2022-08-14T17:27:14.785681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_features = list(sf.get_feature_names_out())\nbest_features","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:27:14.788540Z","iopub.execute_input":"2022-08-14T17:27:14.788862Z","iopub.status.idle":"2022-08-14T17:27:14.796923Z","shell.execute_reply.started":"2022-08-14T17:27:14.788833Z","shell.execute_reply":"2022-08-14T17:27:14.795810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = CatBoostClassifier(verbose=False, eval_metric='Accuracy')\nmodel.fit(X[best_features], y)\nprediction = model.predict(test_df[best_features])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:27:14.798635Z","iopub.execute_input":"2022-08-14T17:27:14.799340Z","iopub.status.idle":"2022-08-14T17:27:18.894737Z","shell.execute_reply.started":"2022-08-14T17:27:14.799308Z","shell.execute_reply":"2022-08-14T17:27:18.893512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Prediction\nfinal = pd.DataFrame()\nfinal.index = test_df.index\nfinal['Transported'] = prediction\nfinal['Transported'].replace(0, False, inplace=True)\nfinal['Transported'].replace(1, True, inplace=True)\nfinal.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T17:27:18.896802Z","iopub.execute_input":"2022-08-14T17:27:18.897145Z","iopub.status.idle":"2022-08-14T17:27:18.913344Z","shell.execute_reply.started":"2022-08-14T17:27:18.897113Z","shell.execute_reply":"2022-08-14T17:27:18.912558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Final score so far: 0.81271 -- in progress</h3>","metadata":{}}]}