{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Part 1: Let's set everything up","metadata":{"papermill":{"duration":0.014782,"end_time":"2022-07-29T20:17:43.211785","exception":false,"start_time":"2022-07-29T20:17:43.197003","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### Importing libraries.","metadata":{"papermill":{"duration":0.016366,"end_time":"2022-07-29T20:17:43.243413","exception":false,"start_time":"2022-07-29T20:17:43.227047","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.185197,"end_time":"2022-07-29T20:17:44.443797","exception":false,"start_time":"2022-07-29T20:17:43.258600","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:24.863815Z","iopub.execute_input":"2022-07-31T09:39:24.864214Z","iopub.status.idle":"2022-07-31T09:39:24.873263Z","shell.execute_reply.started":"2022-07-31T09:39:24.864184Z","shell.execute_reply":"2022-07-31T09:39:24.871867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Importing Data.","metadata":{"papermill":{"duration":0.014982,"end_time":"2022-07-29T20:17:44.476356","exception":false,"start_time":"2022-07-29T20:17:44.461374","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv')\ntest_df = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')\nsample_df = pd.read_csv('/kaggle/input/spaceship-titanic/sample_submission.csv')\n\ntrain = train_df.copy()\ntest = test_df.copy()\nsample = sample_df.copy()\ntrain_test = pd.concat([train,test],axis=0,ignore_index=True)","metadata":{"papermill":{"duration":0.13141,"end_time":"2022-07-29T20:17:44.623505","exception":false,"start_time":"2022-07-29T20:17:44.492095","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:24.875487Z","iopub.execute_input":"2022-07-31T09:39:24.876413Z","iopub.status.idle":"2022-07-31T09:39:24.955600Z","shell.execute_reply.started":"2022-07-31T09:39:24.876366Z","shell.execute_reply":"2022-07-31T09:39:24.954594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Part 2: EDA & Feature Engineering","metadata":{"papermill":{"duration":0.015196,"end_time":"2022-07-29T20:17:44.655063","exception":false,"start_time":"2022-07-29T20:17:44.639867","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### Correlation between some numerical features and the target.","metadata":{"papermill":{"duration":0.014935,"end_time":"2022-07-29T20:17:44.685122","exception":false,"start_time":"2022-07-29T20:17:44.670187","status":"completed"},"tags":[]}},{"cell_type":"code","source":"num_fea_imp = train.corr().Transported.abs().sort_values(ascending=False).drop('Transported',axis=0)\nplt.figure(figsize=(10,5),dpi=100)\nsns.barplot(x=num_fea_imp.index,y=num_fea_imp.values)","metadata":{"papermill":{"duration":0.304932,"end_time":"2022-07-29T20:17:45.005508","exception":false,"start_time":"2022-07-29T20:17:44.700576","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:24.957103Z","iopub.execute_input":"2022-07-31T09:39:24.958179Z","iopub.status.idle":"2022-07-31T09:39:25.202304Z","shell.execute_reply.started":"2022-07-31T09:39:24.958129Z","shell.execute_reply":"2022-07-31T09:39:25.201012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's check out what the traning data looks like.","metadata":{"papermill":{"duration":0.015763,"end_time":"2022-07-29T20:17:45.037321","exception":false,"start_time":"2022-07-29T20:17:45.021558","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train.head()","metadata":{"papermill":{"duration":0.048582,"end_time":"2022-07-29T20:17:45.101674","exception":false,"start_time":"2022-07-29T20:17:45.053092","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.203935Z","iopub.execute_input":"2022-07-31T09:39:25.204310Z","iopub.status.idle":"2022-07-31T09:39:25.227388Z","shell.execute_reply.started":"2022-07-31T09:39:25.204277Z","shell.execute_reply":"2022-07-31T09:39:25.226242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### There are not many features, let's explore them one by one.","metadata":{"papermill":{"duration":0.015996,"end_time":"2022-07-29T20:17:45.133714","exception":false,"start_time":"2022-07-29T20:17:45.117718","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### First, let's check if the target distribution is balanced.","metadata":{"papermill":{"duration":0.01552,"end_time":"2022-07-29T20:17:45.165094","exception":false,"start_time":"2022-07-29T20:17:45.149574","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Transported \ntrain.Transported.sum()/len(train)","metadata":{"papermill":{"duration":0.027752,"end_time":"2022-07-29T20:17:45.209090","exception":false,"start_time":"2022-07-29T20:17:45.181338","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.231052Z","iopub.execute_input":"2022-07-31T09:39:25.231532Z","iopub.status.idle":"2022-07-31T09:39:25.241371Z","shell.execute_reply.started":"2022-07-31T09:39:25.231485Z","shell.execute_reply":"2022-07-31T09:39:25.240182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ok, about half of the passenger were transported. So pretty balanced proportion.\n#### Now, let's see how many values are missing for each feature.","metadata":{"papermill":{"duration":0.016195,"end_time":"2022-07-29T20:17:45.241662","exception":false,"start_time":"2022-07-29T20:17:45.225467","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_test.isnull().sum()","metadata":{"papermill":{"duration":0.04141,"end_time":"2022-07-29T20:17:45.299681","exception":false,"start_time":"2022-07-29T20:17:45.258271","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.242903Z","iopub.execute_input":"2022-07-31T09:39:25.243266Z","iopub.status.idle":"2022-07-31T09:39:25.269635Z","shell.execute_reply.started":"2022-07-31T09:39:25.243227Z","shell.execute_reply":"2022-07-31T09:39:25.268114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Every feature has missing values except for 'PassengerID'.","metadata":{"papermill":{"duration":0.015762,"end_time":"2022-07-29T20:17:45.331836","exception":false,"start_time":"2022-07-29T20:17:45.316074","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### Run this cell below to see if we have overlapping Passenger Groups between train and test data.\n#### Recall that the first 4 digits of the Passenger ID is their group number.\n#### If the result is 0, then no overlapping, meaning that this feature won't probably help too much to the traning process.","metadata":{"papermill":{"duration":0.015883,"end_time":"2022-07-29T20:17:45.364114","exception":false,"start_time":"2022-07-29T20:17:45.348231","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# PaggenserID - Is there overlapping of Passenger Groups between train and test data?\n\ntrain['PassengerGroup'] = train.PassengerId.str.split('_',expand=True)[0].astype('int')\ntest['PassengerGroup'] = test.PassengerId.str.split('_',expand=True)[0].astype('int')\n\nlen(set(train['PassengerGroup'])) + len(set(test['PassengerGroup'])) - len(set(list(train['PassengerGroup'])+list(test['PassengerGroup'])))","metadata":{"papermill":{"duration":0.06545,"end_time":"2022-07-29T20:17:45.445861","exception":false,"start_time":"2022-07-29T20:17:45.380411","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.270954Z","iopub.execute_input":"2022-07-31T09:39:25.271872Z","iopub.status.idle":"2022-07-31T09:39:25.317907Z","shell.execute_reply.started":"2022-07-31T09:39:25.271829Z","shell.execute_reply":"2022-07-31T09:39:25.316743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ok, we will drop the useless features later.\n#### Now move on to the 'HomePlanet' feature.\n#### Let's fill the null values with 'Unknown'.","metadata":{"papermill":{"duration":0.015701,"end_time":"2022-07-29T20:17:45.477776","exception":false,"start_time":"2022-07-29T20:17:45.462075","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# HomePlanet\ntrain['HomePlanet'] = train['HomePlanet'].fillna(value='Unknown')\ntest['HomePlanet'] = test['HomePlanet'].fillna(value='Unknown')","metadata":{"papermill":{"duration":0.030171,"end_time":"2022-07-29T20:17:45.524270","exception":false,"start_time":"2022-07-29T20:17:45.494099","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.319630Z","iopub.execute_input":"2022-07-31T09:39:25.320988Z","iopub.status.idle":"2022-07-31T09:39:25.329206Z","shell.execute_reply.started":"2022-07-31T09:39:25.320948Z","shell.execute_reply":"2022-07-31T09:39:25.328062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now the 'CryoSleep' feature.\n#### The missing values probably indicate that the person is not really dong cryosleep.\n#### Let's just fill them with False.","metadata":{"papermill":{"duration":0.016509,"end_time":"2022-07-29T20:17:45.557023","exception":false,"start_time":"2022-07-29T20:17:45.540514","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# CryoSleep\ntrain['CryoSleep'] = train['CryoSleep'].fillna(value=False)\ntest['CryoSleep'] = test['CryoSleep'].fillna(value=False)","metadata":{"papermill":{"duration":0.031864,"end_time":"2022-07-29T20:17:45.605169","exception":false,"start_time":"2022-07-29T20:17:45.573305","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.330471Z","iopub.execute_input":"2022-07-31T09:39:25.330996Z","iopub.status.idle":"2022-07-31T09:39:25.344195Z","shell.execute_reply.started":"2022-07-31T09:39:25.330960Z","shell.execute_reply":"2022-07-31T09:39:25.342884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### The 'CabinDeck' feature is interesing.\n#### Since it contains location information, let's try to extract what we can.\n#### The 'Cabin' and 'CabinSide' features are created in the cell below.\n#### We didn't include a Cabin Number feature because there are way too many values of it.\n#### Null values are filled with 'U' for unknown.","metadata":{"papermill":{"duration":0.016806,"end_time":"2022-07-29T20:17:45.638259","exception":false,"start_time":"2022-07-29T20:17:45.621453","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Cabin\ntrain['CabinDeck'] = train['Cabin'].str.split('/',expand=True)[0]\ntest['CabinDeck'] = test['Cabin'].str.split('/',expand=True)[0]\ntrain['CabinDeck'] = train['CabinDeck'].fillna(value='U')\ntest['CabinDeck'] = test['CabinDeck'].fillna(value='U')\n\ntrain['CabinSide'] = train['Cabin'].str.split('/',expand=True)[2]\ntest['CabinSide'] = test['Cabin'].str.split('/',expand=True)[2]\ntrain['CabinSide'] = train['CabinSide'].fillna(value='U')\ntest['CabinSide'] = test['CabinSide'].fillna(value='U')","metadata":{"papermill":{"duration":0.086369,"end_time":"2022-07-29T20:17:45.741007","exception":false,"start_time":"2022-07-29T20:17:45.654638","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.345970Z","iopub.execute_input":"2022-07-31T09:39:25.346432Z","iopub.status.idle":"2022-07-31T09:39:25.531793Z","shell.execute_reply.started":"2022-07-31T09:39:25.346401Z","shell.execute_reply":"2022-07-31T09:39:25.530845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now 'Destination'!\n#### First let's map these weird names to some simple letters, because some estimators don't really support rare symbols.\n#### Then we fill the null values with 'U' for unknown.","metadata":{"papermill":{"duration":0.016035,"end_time":"2022-07-29T20:17:45.773612","exception":false,"start_time":"2022-07-29T20:17:45.757577","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Destination\ndest_dic = {'TRAPPIST-1e':'A','55 Cancri e':'B','PSO J318.5-22':'C'}\ntrain['Destination'] = train['Destination'].map(dest_dic)\ntrain['Destination'] = train['Destination'].fillna(value='U')\ntest['Destination'] = test['Destination'].map(dest_dic)\ntest['Destination'] = test['Destination'].fillna(value='U')","metadata":{"papermill":{"duration":0.035593,"end_time":"2022-07-29T20:17:45.825667","exception":false,"start_time":"2022-07-29T20:17:45.790074","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.537514Z","iopub.execute_input":"2022-07-31T09:39:25.537851Z","iopub.status.idle":"2022-07-31T09:39:25.552087Z","shell.execute_reply.started":"2022-07-31T09:39:25.537821Z","shell.execute_reply":"2022-07-31T09:39:25.551063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 'Age' is important.\n#### We can group by 'PassengerGroup' and 'HomePlanet' to fill in the null values.\n#### I tired both mean and median, and it seems the latter one performes better on this dataset.\n#### Let's also create a categorical variable for indicating if the passenger is adult or not.","metadata":{"papermill":{"duration":0.015857,"end_time":"2022-07-29T20:17:45.857972","exception":false,"start_time":"2022-07-29T20:17:45.842115","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Age\ntrain['Age'] = train['Age'].fillna(train.groupby('PassengerGroup')['Age'].transform('median'))\ntrain['Age'] = train['Age'].fillna(train.groupby('HomePlanet')['Age'].transform('median'))\n\ntest['Age'] = test['Age'].fillna(test.groupby('PassengerGroup')['Age'].transform('median'))\ntest['Age'] = test['Age'].fillna(test.groupby('HomePlanet')['Age'].transform('median'))\n\ntrain['Adult'] = 1\ntrain.loc[train['Age']<18, 'Adult'] = 0\n\ntest['Adult'] = 1\ntest.loc[test['Age']<18, 'Adult'] = 0","metadata":{"papermill":{"duration":0.044394,"end_time":"2022-07-29T20:17:45.918693","exception":false,"start_time":"2022-07-29T20:17:45.874299","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.553845Z","iopub.execute_input":"2022-07-31T09:39:25.554163Z","iopub.status.idle":"2022-07-31T09:39:25.574263Z","shell.execute_reply.started":"2022-07-31T09:39:25.554134Z","shell.execute_reply":"2022-07-31T09:39:25.573147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Nothing crazy here with the 'VIP' feature.\n#### It's pretty normal for people who are not VIP to have no records.\n#### So let's just fill in null values with False.","metadata":{"papermill":{"duration":0.016023,"end_time":"2022-07-29T20:17:45.953152","exception":false,"start_time":"2022-07-29T20:17:45.937129","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# VIP\ntrain['VIP'] = train['VIP'].fillna(value=False)\ntest['VIP'] = test['VIP'].fillna(value=False)","metadata":{"papermill":{"duration":0.031848,"end_time":"2022-07-29T20:17:46.001766","exception":false,"start_time":"2022-07-29T20:17:45.969918","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.575544Z","iopub.execute_input":"2022-07-31T09:39:25.576065Z","iopub.status.idle":"2022-07-31T09:39:25.584397Z","shell.execute_reply.started":"2022-07-31T09:39:25.576034Z","shell.execute_reply":"2022-07-31T09:39:25.583630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### The billing features below are numerical values.\n#### Null values probably mean that the passenger did not spend at all (bill = 0).\n#### So let's fill in null values with 0.","metadata":{"papermill":{"duration":0.016031,"end_time":"2022-07-29T20:17:46.034227","exception":false,"start_time":"2022-07-29T20:17:46.018196","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# RoomService, FoodCourt, ShoppingMall, Spa, VRDeck\ntrain[['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']] = train[['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']].fillna(value=0)\ntest[['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']] = test[['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']].fillna(value=0)","metadata":{"papermill":{"duration":0.036155,"end_time":"2022-07-29T20:17:46.086598","exception":false,"start_time":"2022-07-29T20:17:46.050443","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.585526Z","iopub.execute_input":"2022-07-31T09:39:25.586067Z","iopub.status.idle":"2022-07-31T09:39:25.604596Z","shell.execute_reply.started":"2022-07-31T09:39:25.586032Z","shell.execute_reply":"2022-07-31T09:39:25.603438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We can also create a 'TotalSpend' feature to add up all the bill amount.","metadata":{"papermill":{"duration":0.016075,"end_time":"2022-07-29T20:17:46.119233","exception":false,"start_time":"2022-07-29T20:17:46.103158","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Total Spend\ntrain['TotalSpend'] = train['RoomService']+train['FoodCourt']+train['ShoppingMall']+train['Spa']+train['VRDeck']\ntest['TotalSpend'] = test['RoomService']+test['FoodCourt']+test['ShoppingMall']+test['Spa']+test['VRDeck']","metadata":{"papermill":{"duration":0.031386,"end_time":"2022-07-29T20:17:46.166858","exception":false,"start_time":"2022-07-29T20:17:46.135472","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.606243Z","iopub.execute_input":"2022-07-31T09:39:25.607413Z","iopub.status.idle":"2022-07-31T09:39:25.616539Z","shell.execute_reply.started":"2022-07-31T09:39:25.607376Z","shell.execute_reply":"2022-07-31T09:39:25.615247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### There might have something interesing with the 'Name' feature.\n#### We can extract their family name.\n#### Based on their family names, let's create a new feature 'FamilyMember' to count the total of passenger under the same family name.\n#### This might related to the 'Transported' result.","metadata":{"papermill":{"duration":0.015889,"end_time":"2022-07-29T20:17:46.199133","exception":false,"start_time":"2022-07-29T20:17:46.183244","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Name\ntrain['FamilyName'] = train['Name'].str.split(' ',expand=True)[1]\ntrain['FamilyName'] = train['FamilyName'].fillna('Unknown')\n\ntest['FamilyName'] = test['Name'].str.split(' ',expand=True)[1]\ntest['FamilyName'] = test['FamilyName'].fillna('Unknown')\n\ntrain_test['FamilyName'] = train_test['Name'].str.split(' ',expand=True)[1]\ntrain_test['FamilyName'] = train_test['FamilyName'].fillna('Unknown')\n\nfamily_name_dic = train_test['FamilyName'].value_counts().to_dict()\nfamily_name_dic['Unknown'] = 0\n\ntrain['FamilyMember'] = train['FamilyName']\ntrain['FamilyMember'] = train['FamilyMember'].map(family_name_dic)\n\ntest['FamilyMember'] = test['FamilyName']\ntest['FamilyMember'] = test['FamilyMember'].map(family_name_dic)","metadata":{"papermill":{"duration":0.104197,"end_time":"2022-07-29T20:17:46.319560","exception":false,"start_time":"2022-07-29T20:17:46.215363","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.617999Z","iopub.execute_input":"2022-07-31T09:39:25.618815Z","iopub.status.idle":"2022-07-31T09:39:25.698301Z","shell.execute_reply.started":"2022-07-31T09:39:25.618782Z","shell.execute_reply":"2022-07-31T09:39:25.697149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now we drop the extra columns.","metadata":{"papermill":{"duration":0.016422,"end_time":"2022-07-29T20:17:46.352582","exception":false,"start_time":"2022-07-29T20:17:46.336160","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train = train.drop(['PassengerId','PassengerGroup','Cabin','Name','FamilyName'],axis=1)\ntest = test.drop(['PassengerId','PassengerGroup','Cabin','Name','FamilyName'],axis=1)","metadata":{"papermill":{"duration":0.035852,"end_time":"2022-07-29T20:17:46.405082","exception":false,"start_time":"2022-07-29T20:17:46.369230","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.699779Z","iopub.execute_input":"2022-07-31T09:39:25.700677Z","iopub.status.idle":"2022-07-31T09:39:25.712657Z","shell.execute_reply.started":"2022-07-31T09:39:25.700639Z","shell.execute_reply":"2022-07-31T09:39:25.711528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### You will get an error at further steps without converting boolean values to integers (0 and 1).","metadata":{"papermill":{"duration":0.016002,"end_time":"2022-07-29T20:17:46.437450","exception":false,"start_time":"2022-07-29T20:17:46.421448","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Convert Bool to Int\ntrain[['CryoSleep','VIP','Transported']] = train[['CryoSleep','VIP','Transported']].astype(int)\ntest[['CryoSleep','VIP']] = test[['CryoSleep','VIP']].astype(int)","metadata":{"papermill":{"duration":0.033419,"end_time":"2022-07-29T20:17:46.487408","exception":false,"start_time":"2022-07-29T20:17:46.453989","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.714242Z","iopub.execute_input":"2022-07-31T09:39:25.714730Z","iopub.status.idle":"2022-07-31T09:39:25.726314Z","shell.execute_reply.started":"2022-07-31T09:39:25.714655Z","shell.execute_reply":"2022-07-31T09:39:25.725502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now let's check out what we have now.","metadata":{"papermill":{"duration":0.016363,"end_time":"2022-07-29T20:17:46.520045","exception":false,"start_time":"2022-07-29T20:17:46.503682","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train.head()","metadata":{"papermill":{"duration":0.047016,"end_time":"2022-07-29T20:17:46.583989","exception":false,"start_time":"2022-07-29T20:17:46.536973","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.727376Z","iopub.execute_input":"2022-07-31T09:39:25.728452Z","iopub.status.idle":"2022-07-31T09:39:25.752173Z","shell.execute_reply.started":"2022-07-31T09:39:25.728416Z","shell.execute_reply":"2022-07-31T09:39:25.751306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"papermill":{"duration":0.048071,"end_time":"2022-07-29T20:17:46.648771","exception":false,"start_time":"2022-07-29T20:17:46.600700","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.753191Z","iopub.execute_input":"2022-07-31T09:39:25.754283Z","iopub.status.idle":"2022-07-31T09:39:25.782505Z","shell.execute_reply.started":"2022-07-31T09:39:25.754246Z","shell.execute_reply":"2022-07-31T09:39:25.780991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ok, time for feature scaling (numerical features) and encoding (categorical features).","metadata":{"papermill":{"duration":0.016941,"end_time":"2022-07-29T20:17:46.682518","exception":false,"start_time":"2022-07-29T20:17:46.665577","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Feature Scaling\nnum_features = ['Age','RoomService','FoodCourt','ShoppingMall','Spa','VRDeck','TotalSpend','FamilyMember']\n\nfrom sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\ntrain_num_scaled = scaler.fit_transform(train[num_features])\ntest_num_scaled = scaler.transform(test[num_features])\n\ntrain_num_scaled = pd.DataFrame(data=train_num_scaled,columns=num_features)\ntest_num_scaled = pd.DataFrame(data=test_num_scaled,columns=num_features)\n\n# Feature Encoding\ncat_features = ['HomePlanet','CryoSleep','Destination','VIP','CabinDeck','CabinSide','Adult']\n\ntrain_cat_encoded = pd.get_dummies(train[cat_features],drop_first=True)\ntest_cat_encoded = pd.get_dummies(test[cat_features],drop_first=True)","metadata":{"papermill":{"duration":0.195711,"end_time":"2022-07-29T20:17:46.895364","exception":false,"start_time":"2022-07-29T20:17:46.699653","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.784494Z","iopub.execute_input":"2022-07-31T09:39:25.784965Z","iopub.status.idle":"2022-07-31T09:39:25.893076Z","shell.execute_reply.started":"2022-07-31T09:39:25.784917Z","shell.execute_reply":"2022-07-31T09:39:25.891898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We need to concatenate the scaled numerical features and the encoded categorical features.","metadata":{"papermill":{"duration":0.016886,"end_time":"2022-07-29T20:17:46.929386","exception":false,"start_time":"2022-07-29T20:17:46.912500","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Train and Test\nX = pd.concat([train_num_scaled,train_cat_encoded],axis=1)\nX_test = pd.concat([test_num_scaled,test_cat_encoded],axis=1)\ny = train.Transported\nX_all = pd.concat([X,y],axis=1)","metadata":{"papermill":{"duration":0.032864,"end_time":"2022-07-29T20:17:46.979408","exception":false,"start_time":"2022-07-29T20:17:46.946544","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.894462Z","iopub.execute_input":"2022-07-31T09:39:25.894849Z","iopub.status.idle":"2022-07-31T09:39:25.904485Z","shell.execute_reply.started":"2022-07-31T09:39:25.894815Z","shell.execute_reply":"2022-07-31T09:39:25.903498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### After all this, let's check the correlation between the engineered features and the target.","metadata":{"papermill":{"duration":0.01688,"end_time":"2022-07-29T20:17:47.013397","exception":false,"start_time":"2022-07-29T20:17:46.996517","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize=(10,5),dpi=100)\nX_all.corr().Transported.abs().sort_values(ascending=True).iloc[:-1].plot.barh()","metadata":{"papermill":{"duration":0.459002,"end_time":"2022-07-29T20:17:47.489502","exception":false,"start_time":"2022-07-29T20:17:47.030500","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:25.906003Z","iopub.execute_input":"2022-07-31T09:39:25.906810Z","iopub.status.idle":"2022-07-31T09:39:26.285611Z","shell.execute_reply.started":"2022-07-31T09:39:25.906765Z","shell.execute_reply":"2022-07-31T09:39:26.284430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Seems CryoSleep and the bill amounts are higly correlated to the target.\n#### We're ready for modeling now!","metadata":{"papermill":{"duration":0.017625,"end_time":"2022-07-29T20:17:47.524896","exception":false,"start_time":"2022-07-29T20:17:47.507271","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Part 3: Modeling","metadata":{"papermill":{"duration":0.018123,"end_time":"2022-07-29T20:17:47.561279","exception":false,"start_time":"2022-07-29T20:17:47.543156","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### First, let's do a Train Valid Split.\n#### I call it 'Train Valid' instead of 'Train Test' to avoid confusion.\n#### The Valid dataframe will be used for model evaluation below.\n#### The Test dataframe is what we will use for final prediction and submission.","metadata":{"papermill":{"duration":0.017302,"end_time":"2022-07-29T20:17:47.596282","exception":false,"start_time":"2022-07-29T20:17:47.578980","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Train Valid Split\nfrom sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.3, random_state=101)","metadata":{"papermill":{"duration":0.091129,"end_time":"2022-07-29T20:17:47.705213","exception":false,"start_time":"2022-07-29T20:17:47.614084","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:26.286939Z","iopub.execute_input":"2022-07-31T09:39:26.287352Z","iopub.status.idle":"2022-07-31T09:39:26.359413Z","shell.execute_reply.started":"2022-07-31T09:39:26.287320Z","shell.execute_reply":"2022-07-31T09:39:26.358159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's import everthing we will be using.","metadata":{"papermill":{"duration":0.018173,"end_time":"2022-07-29T20:17:47.741421","exception":false,"start_time":"2022-07-29T20:17:47.723248","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nfrom sklearn import metrics\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost.sklearn import XGBClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.ensemble import HistGradientBoostingClassifier\n\nfrom sklearn.model_selection import GridSearchCV","metadata":{"papermill":{"duration":1.701487,"end_time":"2022-07-29T20:17:49.461433","exception":false,"start_time":"2022-07-29T20:17:47.759946","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:26.360923Z","iopub.execute_input":"2022-07-31T09:39:26.361363Z","iopub.status.idle":"2022-07-31T09:39:27.970193Z","shell.execute_reply.started":"2022-07-31T09:39:26.361327Z","shell.execute_reply":"2022-07-31T09:39:27.969024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Run the cell below to find out the best model(s) with their base parameters.","metadata":{"papermill":{"duration":0.017613,"end_time":"2022-07-29T20:17:49.496703","exception":false,"start_time":"2022-07-29T20:17:49.479090","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Logistic Regression\nprint('Running LogisticRegression\\n')\nlogreg = LogisticRegression(max_iter = 600)\nscores = cross_val_score(logreg,X_train,y_train,scoring='neg_mean_squared_error',cv=5)\nlogreg_mse = round(abs(scores.mean()), 4)\nlogreg.fit(X_train, y_train)\ny_pred = logreg.predict(X_valid)\nlogreg_acc = round(metrics.accuracy_score(y_valid, y_pred), 4)\n\n\n# Decision Tree\nprint('Running DecisionTreeClassifier\\n')\ndecision_tree = DecisionTreeClassifier()\nscores = cross_val_score(decision_tree,X_train,y_train,scoring='neg_mean_squared_error',cv=5)\ndecision_tree_mse = round(abs(scores.mean()), 4)\ndecision_tree.fit(X_train, y_train)\ny_pred = decision_tree.predict(X_valid)\ndecision_tree_acc = round(metrics.accuracy_score(y_valid, y_pred), 4)\n\n# Random Forest\nprint('Running RandomForestClassifier\\n')\nrandom_forest = RandomForestClassifier()\nscores = cross_val_score(random_forest,X_train,y_train,scoring='neg_mean_squared_error',cv=5)\nrandom_forest_mse = round(abs(scores.mean()), 4)\nrandom_forest.fit(X_train, y_train)\ny_pred = random_forest.predict(X_valid)\nrandom_forest_acc = round(metrics.accuracy_score(y_valid, y_pred), 4)\n\n# XGBoost\nprint('Running XGBClassifier\\n')\nxgb = XGBClassifier()\nscores = cross_val_score(xgb,X_train,y_train,scoring='neg_mean_squared_error',cv=5)\nxgb_mse = round(abs(scores.mean()), 4)\nxgb.fit(X_train, y_train)\ny_pred = xgb.predict(X_valid)\nxgb_acc = round(metrics.accuracy_score(y_valid, y_pred), 4)\n\n# GB\nprint('Running GradientBoostingClassifier\\n')\ngb = GradientBoostingClassifier()\nscores = cross_val_score(gb,X_train,y_train,scoring='neg_mean_squared_error',cv=5)\ngb_mse = round(abs(scores.mean()), 4)\ngb.fit(X_train, y_train)\ny_pred = gb.predict(X_valid)\ngb_acc = round(metrics.accuracy_score(y_valid, y_pred), 4)\n\n# LightGBM\nprint('Running LGBMClassifier\\n')\nlgbm = LGBMClassifier()\nscores = cross_val_score(lgbm,X_train,y_train,scoring='neg_mean_squared_error',cv=5)\nlgbm_mse = round(abs(scores.mean()), 4)\nlgbm.fit(X_train, y_train)\ny_pred = lgbm.predict(X_valid)\nlgbm_acc = round(metrics.accuracy_score(y_valid, y_pred), 4)\n\n# Catboost\nprint('Running CatBoostClassifier\\n')\ncatb = CatBoostClassifier(verbose = 0)\nscores = cross_val_score(catb,X_train,y_train,scoring='neg_mean_squared_error',cv=5)\ncatb_mse = round(abs(scores.mean()), 4)\ncatb.fit(X_train, y_train)\ny_pred = catb.predict(X_valid)\ncatb_acc = round(metrics.accuracy_score(y_valid, y_pred), 4)\n\n# Histogram-based Gradient Boosting Classification Tree\nprint('Running HistGradientBoostingClassifier\\n')\nhgb = HistGradientBoostingClassifier()\nscores = cross_val_score(hgb,X_train,y_train,scoring='neg_mean_squared_error',cv=5)\nhgb_mse = round(abs(scores.mean()), 4)\nhgb.fit(X_train, y_train)\ny_pred = hgb.predict(X_valid)\nhgb_acc = round(metrics.accuracy_score(y_valid, y_pred), 4)\n\nmodel_df = pd.DataFrame({\n    'Model': ['Logistic Regression', 'Decision Tree', 'Random Forest', 'XGBoost', 'GB', 'LightGBM', 'Catboost', 'HistBoost'],\n    'Train MSE': [logreg_mse, decision_tree_mse, random_forest_mse, xgb_mse, gb_mse, lgbm_mse, catb_mse, hgb_mse],\n    'Validation Accuracy': [logreg_acc, decision_tree_acc, random_forest_acc, xgb_acc, gb_acc, lgbm_acc, catb_acc, hgb_acc]\n})\n\nprint(model_df.sort_values('Validation Accuracy', ascending = False).reset_index(drop = True))","metadata":{"papermill":{"duration":38.566602,"end_time":"2022-07-29T20:18:28.081307","exception":false,"start_time":"2022-07-29T20:17:49.514705","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:39:27.971598Z","iopub.execute_input":"2022-07-31T09:39:27.971956Z","iopub.status.idle":"2022-07-31T09:40:05.946815Z","shell.execute_reply.started":"2022-07-31T09:39:27.971926Z","shell.execute_reply":"2022-07-31T09:40:05.945891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Seems Catboost is one of the best option here with low Train MSE and high Validation Accuracy.\n#### Let's run a Grid Search on it to further tune the hyper parameters.\n#### Note that it might take several minutes to run the cell below.\n#### If you're eager knowing the result, skip to the cell after.","metadata":{"papermill":{"duration":0.018664,"end_time":"2022-07-29T20:18:28.118583","exception":false,"start_time":"2022-07-29T20:18:28.099919","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#  Grid Search on Catboost\ncatb = CatBoostClassifier(verbose = 0)\nparam_grid = {'iterations':[300,400,500,1000],\n              'learning_rate':[0.01,0.03,0.05,0.07,0.09],\n              'depth':[2,5,10]\n             }\n\ngrid = GridSearchCV(estimator=catb, param_grid=param_grid, cv=5)\ngrid.fit(X,y)\nprint('Mean accuracy:',grid.score(X,y))\nprint('Best hyperparameters:',grid.best_params_)","metadata":{"papermill":{"duration":1254.373206,"end_time":"2022-07-29T20:39:22.511481","exception":false,"start_time":"2022-07-29T20:18:28.138275","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T09:40:05.950978Z","iopub.execute_input":"2022-07-31T09:40:05.952931Z","iopub.status.idle":"2022-07-31T10:00:31.879973Z","shell.execute_reply.started":"2022-07-31T09:40:05.952880Z","shell.execute_reply":"2022-07-31T10:00:31.878776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Here we instantiate a Catboost model with tuned hyper parameters","metadata":{"papermill":{"duration":0.018349,"end_time":"2022-07-29T20:39:22.548409","exception":false,"start_time":"2022-07-29T20:39:22.530060","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Tuned Catboost\ncatb = CatBoostClassifier(depth=5,iterations=300,learning_rate=0.05,verbose=0)","metadata":{"papermill":{"duration":0.028483,"end_time":"2022-07-29T20:39:22.595532","exception":false,"start_time":"2022-07-29T20:39:22.567049","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T10:00:31.881753Z","iopub.execute_input":"2022-07-31T10:00:31.882175Z","iopub.status.idle":"2022-07-31T10:00:31.889094Z","shell.execute_reply.started":"2022-07-31T10:00:31.882131Z","shell.execute_reply":"2022-07-31T10:00:31.887299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's make a prediction and submit our result!","metadata":{"papermill":{"duration":0.018742,"end_time":"2022-07-29T20:39:22.633402","exception":false,"start_time":"2022-07-29T20:39:22.614660","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Submitting\nmodel=catb\nmodel.fit(X,y)\npredictions = model.predict(X_test)\noutput = pd.DataFrame({'PassengerId': test_df.PassengerId, 'Transported': predictions})\noutput['Transported'] = output['Transported'].astype('bool')\noutput.to_csv('submission.csv', index=False)","metadata":{"papermill":{"duration":1.028111,"end_time":"2022-07-29T20:39:23.680145","exception":false,"start_time":"2022-07-29T20:39:22.652034","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-31T10:00:31.895332Z","iopub.execute_input":"2022-07-31T10:00:31.895731Z","iopub.status.idle":"2022-07-31T10:00:32.907730Z","shell.execute_reply.started":"2022-07-31T10:00:31.895669Z","shell.execute_reply":"2022-07-31T10:00:32.906141Z"},"trusted":true},"execution_count":null,"outputs":[]}]}