{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Contents\n\n\n1. **Exploratory data analysis**\n    - With notes and insights.\n2. **Dealing with Missing values**\n3. **Feature engineering**\n    - Where I use the information provided by the EDA and transform the data to obtain more useful features. \n    - Where I apply preprocessing techniques.\n4. **Model selection**\n    1. Perform hyperparameter tuning\n    2. Select 2 best models\n    3. Use cross validation to select the final best model","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import GradientBoostingClassifier, RandomForestClassifier\nfrom sklearn.model_selection import GridSearchCV \nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import train_test_split\n\n\n\nimport numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nsns.set(style='darkgrid')\n\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T21:39:25.911883Z","iopub.execute_input":"2022-08-08T21:39:25.913456Z","iopub.status.idle":"2022-08-08T21:39:25.925231Z","shell.execute_reply.started":"2022-08-08T21:39:25.913408Z","shell.execute_reply":"2022-08-08T21:39:25.923909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory data analysis\n\nNote: This part was inspired by the notebook of *Samuel Cortinhas*.","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/spaceship-titanic/train.csv')\ntest = pd.read_csv('../input/spaceship-titanic/test.csv')\n\nprint(f\"Train test shape: {train.shape}\")\nprint(f\"Test test shape: {test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:53.041762Z","iopub.execute_input":"2022-08-08T20:55:53.042373Z","iopub.status.idle":"2022-08-08T20:55:53.090190Z","shell.execute_reply.started":"2022-08-08T20:55:53.042335Z","shell.execute_reply":"2022-08-08T20:55:53.088880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:53.244596Z","iopub.execute_input":"2022-08-08T20:55:53.244985Z","iopub.status.idle":"2022-08-08T20:55:53.281706Z","shell.execute_reply.started":"2022-08-08T20:55:53.244954Z","shell.execute_reply":"2022-08-08T20:55:53.280392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:53.443683Z","iopub.execute_input":"2022-08-08T20:55:53.444830Z","iopub.status.idle":"2022-08-08T20:55:53.465864Z","shell.execute_reply.started":"2022-08-08T20:55:53.444779Z","shell.execute_reply":"2022-08-08T20:55:53.465015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:53.642704Z","iopub.execute_input":"2022-08-08T20:55:53.643189Z","iopub.status.idle":"2022-08-08T20:55:53.659181Z","shell.execute_reply.started":"2022-08-08T20:55:53.643114Z","shell.execute_reply":"2022-08-08T20:55:53.657763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Countplot\nsns.countplot(x='Transported', data=train)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:53.840761Z","iopub.execute_input":"2022-08-08T20:55:53.841522Z","iopub.status.idle":"2022-08-08T20:55:54.028278Z","shell.execute_reply.started":"2022-08-08T20:55:53.841479Z","shell.execute_reply":"2022-08-08T20:55:54.027146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The target is balanced so we have not to use imbalanced dataset strategies.","metadata":{"execution":{"iopub.status.busy":"2022-07-27T23:52:21.527783Z","iopub.execute_input":"2022-07-27T23:52:21.528534Z","iopub.status.idle":"2022-07-27T23:52:21.535137Z","shell.execute_reply.started":"2022-07-27T23:52:21.528494Z","shell.execute_reply":"2022-07-27T23:52:21.534071Z"}}},{"cell_type":"markdown","source":"## Checking categorical variables","metadata":{"execution":{"iopub.status.busy":"2022-07-27T23:57:51.130359Z","iopub.execute_input":"2022-07-27T23:57:51.130748Z","iopub.status.idle":"2022-07-27T23:57:51.171441Z","shell.execute_reply.started":"2022-07-27T23:57:51.130717Z","shell.execute_reply":"2022-07-27T23:57:51.170670Z"}}},{"cell_type":"code","source":"train.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:54.039070Z","iopub.execute_input":"2022-08-08T20:55:54.040302Z","iopub.status.idle":"2022-08-08T20:55:54.064870Z","shell.execute_reply.started":"2022-08-08T20:55:54.040250Z","shell.execute_reply":"2022-08-08T20:55:54.063842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = ['HomePlanet', 'CryoSleep', 'Destination', 'VIP']\n\n# Countplot \nfig=plt.figure(figsize=(10,16))\nfor i, feature in enumerate(cat_features):\n    ax=fig.add_subplot(4,2,i+1)\n    sns.countplot(data=train, x=feature, axes=ax, hue='Transported')\n    ax.set_title(feature)\nfig.tight_layout()  # Improves appearance \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:54.231817Z","iopub.execute_input":"2022-08-08T20:55:54.232593Z","iopub.status.idle":"2022-08-08T20:55:55.007314Z","shell.execute_reply.started":"2022-08-08T20:55:54.232548Z","shell.execute_reply":"2022-08-08T20:55:55.006162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Notes**:\n\n- Homeplanet, CyroSleep appears to be good features because there is variation between them.\n- VIP does not appear to be good feature because almost all passanger were not VIP, and futhermore, there is not much difference about the numbers whose have been transported or not. \n- The `PSO J318.5-22` destination, also appears to not be a good feature.\n\n**Insights**:\n- I will delete VIP feature.","metadata":{}},{"cell_type":"markdown","source":"## Checking continous variables","metadata":{}},{"cell_type":"code","source":"# Age distribution\n# Histogram\nplt.figure(figsize=(10,5))\nsns.histplot(data=train, x='Age', hue='Transported', kde=True)\nplt.title('Age distribution')\nplt.xlabel('Age (years)')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:55.009297Z","iopub.execute_input":"2022-08-08T20:55:55.009665Z","iopub.status.idle":"2022-08-08T20:55:55.544163Z","shell.execute_reply.started":"2022-08-08T20:55:55.009617Z","shell.execute_reply":"2022-08-08T20:55:55.543178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Notes:\n\n- 0-18 year olds have more chance to **be transported**.\n- 18-25 year olds have more chance to **not be transported**.\n- Over 25 year olds have **equally chance** to be transported.\n\nInsight:\n\nCreate a new feature that indicates whether the passanger is:\n\n* Child (0-12 years old).\n* Adolescent (12-18 years old). \n* Adult (19-26)\n* Adult (27-35)\n* Adult (36-42)\n* Adult (42+)","metadata":{}},{"cell_type":"code","source":"# Expenditure distribution\n# Histogram\nexpend_cols =['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\n\nfig=plt.figure(figsize=(10,16))\n\n# Plot expenditure features\nfig=plt.figure(figsize=(10,20))\nfor i, feature in enumerate(expend_cols):\n    # Left plot\n    ax=fig.add_subplot(5,2,2*i+1)\n    sns.histplot(data=train, x=feature, axes=ax, bins=40, kde=False, hue='Transported')\n    ax.set_title(feature)\n    \n    # Right plot (truncated)\n    ax=fig.add_subplot(5,2,2*i+2)\n    sns.histplot(data=train, x=feature, axes=ax, bins=40, kde=True, hue='Transported')\n    plt.ylim([0,100])\n    ax.set_title(f\"A zoom in {feature}\")\nfig.tight_layout()  # Improves appearance a bit\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:55.545875Z","iopub.execute_input":"2022-08-08T20:55:55.546277Z","iopub.status.idle":"2022-08-08T20:55:59.885534Z","shell.execute_reply.started":"2022-08-08T20:55:55.546244Z","shell.execute_reply":"2022-08-08T20:55:59.884649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Notes**:\n\n- Most people don't spend any money (as we can see on the left).\n- The distribution of spending decays exponentially and are highly skewed (as we can see on the right).\n- There are a small number of outliers.\n- People who were transported tended to spend less.\n\n**Insight**:\n\n- Create a new feature that tracks the total expenditure across all 5 amenities.\n- Create a binary feature to indicate if the person has not spent anything. (i.e. total expenditure is 0).\n- Log the expenditure features to reduce skewess.","metadata":{}},{"cell_type":"markdown","source":"## Qualitative features","metadata":{}},{"cell_type":"markdown","source":"**Cabin feature**\n\nNow let's take a look at `Cabin`. From this column I can extract the `deck/num/side` from each entry and see how them relationates with `Transported`","metadata":{}},{"cell_type":"code","source":"train['Cabin'].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:59.888296Z","iopub.execute_input":"2022-08-08T20:55:59.888701Z","iopub.status.idle":"2022-08-08T20:55:59.896535Z","shell.execute_reply.started":"2022-08-08T20:55:59.888665Z","shell.execute_reply":"2022-08-08T20:55:59.895603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Cabin_Deck'] = train['Cabin'].str[0]\ntrain['Cabin_Num'] = train['Cabin'].str[2]\ntrain['Cabin_Side'] = train['Cabin'].str[-1]\n\nnew_features = ['Cabin_Deck', 'Cabin_Num', 'Cabin_Side']\n\n# Countplot \nfig=plt.figure(figsize=(10,16))\nfor i, feature in enumerate(new_features):\n    ax=fig.add_subplot(3,1,i+1)\n    sns.countplot(data=train, x=feature, axes=ax, hue='Transported')\n    ax.set_title(feature)\nfig.tight_layout()  # Improves appearance \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:55:59.897961Z","iopub.execute_input":"2022-08-08T20:55:59.898398Z","iopub.status.idle":"2022-08-08T20:56:00.711796Z","shell.execute_reply.started":"2022-08-08T20:55:59.898363Z","shell.execute_reply":"2022-08-08T20:56:00.710339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Notes**:\n\n- The passengers in cabin P are less likely to be transported.\n- The passengers in cabin S are more likely to be transported.\n\n**Insights**:\n\n- Cabin deck `T` and cabin number `0` seems to be outliers and I will remove them.","metadata":{}},{"cell_type":"markdown","source":"**Passenger ID**\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T16:37:20.928276Z","iopub.execute_input":"2022-07-30T16:37:20.928672Z","iopub.status.idle":"2022-07-30T16:37:20.935486Z","shell.execute_reply.started":"2022-07-30T16:37:20.928641Z","shell.execute_reply":"2022-07-30T16:37:20.933683Z"}}},{"cell_type":"code","source":"train['PassengerId'].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:00.713649Z","iopub.execute_input":"2022-08-08T20:56:00.714079Z","iopub.status.idle":"2022-08-08T20:56:00.721256Z","shell.execute_reply.started":"2022-08-08T20:56:00.714043Z","shell.execute_reply":"2022-08-08T20:56:00.720465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Notes**: \n\nPassengerId takes the form gggg_pp where gggg indicates a group the passenger is travelling with and pp is their number within the group.\n\n**Insight**:\n\nFrom `PassengerId` I will extract the group of a passenger and how many persons are in that group. ","metadata":{}},{"cell_type":"code","source":"train['Groups'] = train['PassengerId'].str.slice(start=0, stop=4)\ntrain['Groups'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:00.722268Z","iopub.execute_input":"2022-08-08T20:56:00.723159Z","iopub.status.idle":"2022-08-08T20:56:00.744739Z","shell.execute_reply.started":"2022-08-08T20:56:00.723105Z","shell.execute_reply":"2022-08-08T20:56:00.743513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Notes**:\n\nThere are a high number of groups\n\n**Insight**:\n    \nFor high cardinality, performing one hot encoding in this feature will unpractical, so this feature (`groups`) will not be considered. \n","metadata":{}},{"cell_type":"code","source":"train['member_group'] = train['PassengerId'].str.slice(start=5, stop=7)\ntrain['member_group'] = train['member_group'].astype(int)\n\n# Plot member groups\nplt.figure(figsize=(12,6))\nsns.countplot(data=train, x='member_group', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:00.746541Z","iopub.execute_input":"2022-08-08T20:56:00.747666Z","iopub.status.idle":"2022-08-08T20:56:01.059234Z","shell.execute_reply.started":"2022-08-08T20:56:00.747623Z","shell.execute_reply":"2022-08-08T20:56:01.058003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Notes**:\n\n- There was a high number of passenger which are traveled alone.\n\n**Insight**:\n- Create a dummy variable to indicate if the passenger was traveled alone. ","metadata":{}},{"cell_type":"markdown","source":"# Missing values","metadata":{}},{"cell_type":"code","source":"# see percentages of missing values\n(train.isna().sum()/train.shape[0]) * 100","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:01.060753Z","iopub.execute_input":"2022-08-08T20:56:01.061156Z","iopub.status.idle":"2022-08-08T20:56:01.079934Z","shell.execute_reply.started":"2022-08-08T20:56:01.061104Z","shell.execute_reply":"2022-08-08T20:56:01.078740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Notes**\n\nFeatures has less than 5% of NaN values, which is a good sign. For numerical features, is a good practice to impute missing values with the median for columns whose has a percentage of missing values less than the 5% (which is the case) and for categorical values, replace the missing values with the mode. \n\n\n**Insight**\n\nReplace missing numerical values with the median and missing categorical values with the mode.","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:20:52.110361Z","iopub.execute_input":"2022-07-30T17:20:52.110773Z","iopub.status.idle":"2022-07-30T17:20:52.132779Z","shell.execute_reply.started":"2022-07-30T17:20:52.110742Z","shell.execute_reply":"2022-07-30T17:20:52.131765Z"}}},{"cell_type":"code","source":"cat_features = ['HomePlanet', 'CryoSleep', 'Destination', 'VIP', \n                'Cabin_Deck', 'Cabin_Num', 'Cabin_Side']\n\nnum_features = ['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', \n                'Spa', 'VRDeck']\n\n# replace nan values with the mode for categorical features\nfor feature in cat_features:\n    mode = train[feature].mode()[0]\n    train[feature].fillna(mode, inplace=True)\n    \n# replace nan values with the median for numerical features\nfor feature in num_features:\n    median = train[feature].median()\n    train[feature].fillna(median, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:01.083671Z","iopub.execute_input":"2022-08-08T20:56:01.084023Z","iopub.status.idle":"2022-08-08T20:56:01.110668Z","shell.execute_reply.started":"2022-08-08T20:56:01.083991Z","shell.execute_reply":"2022-08-08T20:56:01.109577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# see percentages of missing values\n(train.isna().sum()/train.shape[0]) * 100","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:01.112093Z","iopub.execute_input":"2022-08-08T20:56:01.112807Z","iopub.status.idle":"2022-08-08T20:56:01.131002Z","shell.execute_reply.started":"2022-08-08T20:56:01.112760Z","shell.execute_reply":"2022-08-08T20:56:01.130187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Notes**: \n\n\nCabin feature will not be used and then it will be removed to avoid multicolinearity with `Cabin_Deck`, `Cabin_Num` and `Cabin_Side`.\n\nName feature will not be used neither because I will use `Passenger_id` as the only identifier for each passenger, which is unique. ","metadata":{}},{"cell_type":"markdown","source":"# Feature engineering\n\nNow I will take all the *insights* and perform the appropiate transformations.","metadata":{}},{"cell_type":"markdown","source":"**Age groups**","metadata":{}},{"cell_type":"code","source":"## Age groups\ntrain['Age_group'] = np.nan\ntrain.loc[train['Age'].between(0,12, inclusive='both'),'Age_group']='Age_0-12'\ntrain.loc[train['Age'].between(13,18, inclusive='both'),'Age_group']='Age_13-18'\ntrain.loc[train['Age'].between(19,26, inclusive='both'),'Age_group']='Age_19-26'\ntrain.loc[train['Age'].between(27,35, inclusive='both'),'Age_group']='Age_27-35'\ntrain.loc[train['Age'].between(36,42, inclusive='both'),'Age_group']='Age_36-42'\ntrain.loc[train['Age'].between(43,50, inclusive='both'),'Age_group']='Age_43-50'\ntrain.loc[train['Age'] > 51,'Age_group']='Age_51+'\n\n# Visualize age groups\nfig=plt.figure(figsize=(10,4))\nsns.countplot(data=train, x='Age_group', hue='Transported',\n             order=['Age_0-12', 'Age_13-18', 'Age_19-26', 'Age_27-35',\n                    'Age_36-42', 'Age_43-50', 'Age_51+'])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:01.132383Z","iopub.execute_input":"2022-08-08T20:56:01.132718Z","iopub.status.idle":"2022-08-08T20:56:01.436236Z","shell.execute_reply.started":"2022-08-08T20:56:01.132688Z","shell.execute_reply":"2022-08-08T20:56:01.435317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Total expenditure \nexpend_cols =['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\ntrain['Total_expenditure']  = train[expend_cols].sum(axis=1)\n\n# Passenger had expenses\ntrain['Had_expenses'] = np.nan\ntrain.loc[train['Total_expenditure'] == 0, 'Had_expenses'] = 'No'\ntrain.loc[train['Total_expenditure'] > 0, 'Had_expenses'] = 'Yes'\n\n\n# Visualization\n# Right plot: total expenditure\nfig=plt.figure(figsize=(12,4))\nplt.subplot(1,2,1)\nsns.histplot(data=train, x='Total_expenditure', hue='Transported', bins=200, kde=True)\nplt.title('Total expenditure')\nplt.ylim([0,150])\nplt.xlim([0,15000])\n\n# Left plot: Expending indicator\nplt.subplot(1,2,2)\nsns.countplot(data=train, x='Had_expenses', hue='Transported')\nplt.title('Passenger had expenses?')\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:01.437479Z","iopub.execute_input":"2022-08-08T20:56:01.437800Z","iopub.status.idle":"2022-08-08T20:56:02.936557Z","shell.execute_reply.started":"2022-08-08T20:56:01.437771Z","shell.execute_reply":"2022-08-08T20:56:02.935394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Note**\n\nTo logarize the expenditure cols, we must remember that the **majority of passengers have a total expenditure of zero**. The log of zero is **indetermined** so in this case we must add `1` to the expenses in order to use the log function.","metadata":{}},{"cell_type":"code","source":"expend_cols = expend_cols + ['Total_expenditure']\n\n# Plot log transform results\nfig=plt.figure(figsize=(12,20))\nfor i, col in enumerate(expend_cols):\n    plt.subplot(6,2,2*i+1)\n    sns.histplot(train[col], binwidth=100)\n    plt.ylim([0,200])\n    plt.title(f'{col} (original)')\n    \n    plt.subplot(6,2,2*i+2)\n    sns.histplot(np.log(1 + train[col]))\n    plt.ylim([0,200])\n    plt.title(f'{col} (log-transform)')\n    \nfig.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:02.938316Z","iopub.execute_input":"2022-08-08T20:56:02.938646Z","iopub.status.idle":"2022-08-08T20:56:08.913849Z","shell.execute_reply.started":"2022-08-08T20:56:02.938616Z","shell.execute_reply":"2022-08-08T20:56:08.912440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply log-transformation\nfor col in expend_cols:\n    train[col] = np.log(1+train[col])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:08.915796Z","iopub.execute_input":"2022-08-08T20:56:08.916885Z","iopub.status.idle":"2022-08-08T20:56:08.926336Z","shell.execute_reply.started":"2022-08-08T20:56:08.916842Z","shell.execute_reply":"2022-08-08T20:56:08.924970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Alone passenger**","metadata":{}},{"cell_type":"code","source":"# Alone passenger\ntrain['Alone'] = np.where(train['member_group'] == 1, 'Yes','No')\n\n# Visualize\nfig=plt.figure(figsize=(10,4))\nsns.countplot(data=train, x='Alone', hue='Transported')\nplt.title('Passenger traveled alone')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:08.927819Z","iopub.execute_input":"2022-08-08T20:56:08.928170Z","iopub.status.idle":"2022-08-08T20:56:09.166832Z","shell.execute_reply.started":"2022-08-08T20:56:08.928124Z","shell.execute_reply":"2022-08-08T20:56:09.165729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Cabin_Deck'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:09.169417Z","iopub.execute_input":"2022-08-08T20:56:09.170123Z","iopub.status.idle":"2022-08-08T20:56:09.179898Z","shell.execute_reply.started":"2022-08-08T20:56:09.170078Z","shell.execute_reply":"2022-08-08T20:56:09.178789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove outliers\ntrain = train.query('Cabin_Deck != \"T\"')\ntrain = train.query('Cabin_Num != \"0\"')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:09.181561Z","iopub.execute_input":"2022-08-08T20:56:09.181881Z","iopub.status.idle":"2022-08-08T20:56:09.205417Z","shell.execute_reply.started":"2022-08-08T20:56:09.181853Z","shell.execute_reply":"2022-08-08T20:56:09.204558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Change CyroSleep to Yes or No (to one hot encoding later)\ntrain['CryoSleep'] = train['CryoSleep'].replace({True: 'Yes', False: 'No'})","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:09.206758Z","iopub.execute_input":"2022-08-08T20:56:09.207321Z","iopub.status.idle":"2022-08-08T20:56:09.215434Z","shell.execute_reply.started":"2022-08-08T20:56:09.207288Z","shell.execute_reply":"2022-08-08T20:56:09.214258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:09.218863Z","iopub.execute_input":"2022-08-08T20:56:09.219296Z","iopub.status.idle":"2022-08-08T20:56:09.227418Z","shell.execute_reply.started":"2022-08-08T20:56:09.219249Z","shell.execute_reply":"2022-08-08T20:56:09.226344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test set**","metadata":{}},{"cell_type":"code","source":"## Cabin features\ntest['Cabin_Deck'] = test['Cabin'].str[0]\ntest['Cabin_Num'] = test['Cabin'].str[2]\ntest['Cabin_Side'] = test['Cabin'].str[-1]\n\n## Alone passenger\ntest['member_group'] = test['PassengerId'].str.slice(start=5, stop=7)\ntest['member_group'] = test['member_group'].astype(int)\ntest['Alone'] = np.where(test['member_group'] == 1, 'Yes','No')\n\n## Age groups\ntest['Age_group'] = np.nan\ntest.loc[test['Age'].between(0,12, inclusive='both'),'Age_group']='Age_0-12'\ntest.loc[test['Age'].between(13,18, inclusive='both'),'Age_group']='Age_13-18'\ntest.loc[test['Age'].between(19,26, inclusive='both'),'Age_group']='Age_19-26'\ntest.loc[test['Age'].between(27,35, inclusive='both'),'Age_group']='Age_27-35'\ntest.loc[test['Age'].between(36,42, inclusive='both'),'Age_group']='Age_36-42'\ntest.loc[test['Age'].between(43,50, inclusive='both'),'Age_group']='Age_43-50'\ntest.loc[test['Age'] > 51,'Age_group']='Age_51+'\n\n## Total expenditure \nexpend_cols =['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\ntest['Total_expenditure']  = test[expend_cols].sum(axis=1)\n\n## Log transformation\nexpend_cols = expend_cols + ['Total_expenditure']\nfor col in expend_cols:\n    test[col] = np.log(1+test[col])\n    \n## Passenger had expenses\ntest['Had_expenses'] = np.nan\ntest.loc[test['Total_expenditure'] == 0, 'Had_expenses'] = 'No'\ntest.loc[test['Total_expenditure'] > 0, 'Had_expenses'] = 'Yes'\n\n## Remove outliers\ntest = test.query('Cabin_Deck != \"T\"')\ntest = test.query('Cabin_Num != \"0\"')\n\n## Change Cyrosleep to Yes or No\ntest['CryoSleep'] = test['CryoSleep'].replace({True: 'Yes', False: 'No'})","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:09.228920Z","iopub.execute_input":"2022-08-08T20:56:09.229589Z","iopub.status.idle":"2022-08-08T20:56:09.292048Z","shell.execute_reply.started":"2022-08-08T20:56:09.229540Z","shell.execute_reply":"2022-08-08T20:56:09.291117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop qualitative/redundant/collinear/high cardinality features\ndrop_test_columns = ['Cabin', 'Age', 'VIP', 'Name', 'member_group']\ndrop_train_columns = drop_test_columns + ['PassengerId', 'Groups']\n\ntrain.drop(drop_train_columns, axis=1, inplace=True)\ntest.drop(drop_test_columns, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:09.293366Z","iopub.execute_input":"2022-08-08T20:56:09.293900Z","iopub.status.idle":"2022-08-08T20:56:09.305698Z","shell.execute_reply.started":"2022-08-08T20:56:09.293867Z","shell.execute_reply":"2022-08-08T20:56:09.304681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing techniques\n\n- Imputer: median for numerical values and mode for categorical values.\n- One Hot Encoding for categorical values\n- Standarization for numerical values","metadata":{}},{"cell_type":"code","source":"X = train.drop('Transported', axis=1)\ny = train['Transported']\n\nX_test = test.drop('PassengerId', axis=1)\n\nprint(\"X shape: \", X.shape)\nprint(\"y shape: \", y.shape)\nprint(\"X_test shape: \", X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T20:56:09.311036Z","iopub.execute_input":"2022-08-08T20:56:09.311749Z","iopub.status.idle":"2022-08-08T20:56:09.322680Z","shell.execute_reply.started":"2022-08-08T20:56:09.311713Z","shell.execute_reply":"2022-08-08T20:56:09.321461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Identify categori.absal and numerical features\ncat_features = [col for col in train.columns if train[col].dtypes == 'object']\nnum_features = [col for col in train.columns if train[col].dtypes in ['int64', 'float64']]\n\n\nX_train, X_val, y_train, y_val = train_test_split(X,y, test_size=0.3,\n                                                 stratify=y,\n                                                random_state=42)\n\nnum_steps = [(\"median_imputer\", SimpleImputer(strategy='median')),\n            (\"scaler\", StandardScaler())]\n             \ncat_steps = [(\"mode_imputer\", SimpleImputer(strategy='most_frequent')),\n            (\"ohe\", OneHotEncoder())]\n\nnumerical_transformer = Pipeline(steps=num_steps)\ncategorical_transformer = Pipeline(steps=cat_steps)\n\n# Combine preprocessing\nct = ColumnTransformer(\n    transformers=[\n        ('cat', categorical_transformer, cat_features),\n        ('num', numerical_transformer, num_features),],\n        remainder='passthrough')\n\n\n# Apply preprocessing\nX_train = ct.fit_transform(X_train)\nX_val = ct.transform(X_val)\n\n# Print new shape\nprint('Training set shape: ', X_train.shape)\nprint('Val set shape: ', X_val.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T21:06:03.263152Z","iopub.execute_input":"2022-08-08T21:06:03.263536Z","iopub.status.idle":"2022-08-08T21:06:03.350328Z","shell.execute_reply.started":"2022-08-08T21:06:03.263505Z","shell.execute_reply":"2022-08-08T21:06:03.349008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model selection\n\nI will use the following algorithms: \n\n**Logistic Regression (LR)**: Unlike linear regression which uses Least Squares, this model uses Maximum Likelihood Estimation to fit a sigmoid-curve on the target variable distribution. The sigmoid/logistic curve is commonly used when the data is questions had binary output.\n\n**K-Nearest Neighbors (KNN)**: KNN works by selecting the majority class of the k-nearest neighbours, where the metric used is usually Euclidean distance. It is a simple and effective algorithm but can be sensitive by many factors, e.g. the value of k, the preprocessing done to the data and the metric used.\n\n\n**Random Forest (RF)**: RF is a reliable ensemble of decision trees, which can be used for regression or classification problems. Here, the individual trees are built via bagging (i.e. aggregation of bootstraps which are nothing but multiple train datasets created via sampling with replacement) and split using fewer features. The resulting diverse forest of uncorrelated trees exhibits reduced variance; therefore, is more robust towards change in data and carries its prediction accuracy to new data. It works well with both continuous & categorical data.\n\n\n**Stochastic Gradient Boosting (SGB)**:","metadata":{}},{"cell_type":"markdown","source":"## Hyperparameter tuning","metadata":{}},{"cell_type":"code","source":"# Classifiers\nmodels = {\n    \"LogisticRegression\" : LogisticRegression(random_state=42, solver='liblinear'),\n    \"KNN\" : KNeighborsClassifier(),\n    \"RandomForest\" : RandomForestClassifier(random_state=42),\n    \"GradientBoosting\": GradientBoostingClassifier(random_state=42)\n}\n\n\n# Grids for grid search\nLR_grid = {'penalty': ['l1', 'l2'],\n           'C': [100, 10, 1.0, 0.1, 0.01]}\n\nKNN_grid = {'n_neighbors': [3, 5, 7, 9, 12]}\n\n\nRF_grid = {'n_estimators': [50, 100, 150, 200, 250, 300],\n        'max_depth': [4, 6, 8, 10, 12]}\n\nGB_grid = {'n_estimators': [50, 100, 150, 200, 250, 300],\n        'max_depth': [4, 6, 8, 10, 12]}\n\n# Dictionary of all grids\ngrid = {\n    \"LogisticRegression\" : LR_grid,\n    \"KNN\" : KNN_grid,\n    \"RandomForest\" : RF_grid,\n    \"GradientBoosting\" : GB_grid\n        }","metadata":{"execution":{"iopub.status.busy":"2022-08-08T21:21:24.103996Z","iopub.execute_input":"2022-08-08T21:21:24.104490Z","iopub.status.idle":"2022-08-08T21:21:24.115011Z","shell.execute_reply.started":"2022-08-08T21:21:24.104450Z","shell.execute_reply":"2022-08-08T21:21:24.113733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_scores= pd.DataFrame({'Classifer':models.keys(), \n                            'Validation accuracy': np.zeros(len(models)),\n                           'Training time (min)': np.zeros(len(models))})","metadata":{"execution":{"iopub.status.busy":"2022-08-08T21:41:46.597243Z","iopub.execute_input":"2022-08-08T21:41:46.597762Z","iopub.status.idle":"2022-08-08T21:41:46.607763Z","shell.execute_reply.started":"2022-08-08T21:41:46.597723Z","shell.execute_reply":"2022-08-08T21:41:46.606298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i=0\nmodel_best_params=models.copy()\nfor key, model in models.items():\n    start = time.time()\n    clf = GridSearchCV(estimator=model, param_grid=grid[key], n_jobs=-1, cv=None)\n\n    # Train and score\n    print(f\"Training {model}\")\n    clf.fit(X_train, y_train)\n    valid_scores.iloc[i,1]=clf.score(X_val, y_val)\n    \n    # Save trained model\n    print(f\"Saving {model} best params\")\n    model_best_params[key]=clf.best_params_\n    \n    # Save traiing time\n    stop = time.time()\n    valid_scores.iloc[i,2]=np.round((stop - start)/60, 2)\n    print(f\"Training time (mins): {valid_scores.iloc[i,2]}\\n\")\n    i+=1\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-08T21:41:54.137062Z","iopub.execute_input":"2022-08-08T21:41:54.138347Z","iopub.status.idle":"2022-08-08T21:47:36.081997Z","shell.execute_reply.started":"2022-08-08T21:41:54.138303Z","shell.execute_reply":"2022-08-08T21:47:36.080513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show results\nvalid_scores","metadata":{"execution":{"iopub.status.busy":"2022-08-08T21:47:40.923740Z","iopub.execute_input":"2022-08-08T21:47:40.924200Z","iopub.status.idle":"2022-08-08T21:47:40.938199Z","shell.execute_reply.started":"2022-08-08T21:47:40.924157Z","shell.execute_reply":"2022-08-08T21:47:40.937198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Motivated by this, I will take GradientBoosting and RandomForest to the final stage.\n\n","metadata":{}}]}