{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T16:04:47.348006Z","iopub.execute_input":"2022-08-01T16:04:47.348508Z","iopub.status.idle":"2022-08-01T16:04:47.361929Z","shell.execute_reply.started":"2022-08-01T16:04:47.348464Z","shell.execute_reply":"2022-08-01T16:04:47.360617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Introduction\n\n#### We will cover:\n\n#### 1. Exploratory Data Analysis (EDA)\n#### 2. Feature Engineering\n-     Missing Value\n-     Outliers Analysis\n-     Feature Create\n-     Encoding\n-     Feature Scaling\n-     Feature Selection\n\n#### 3. Modeling\n#### 4. Ensembling","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.express as px\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True) # Avoid the notebook cannot show plotly\n\ndf_train = pd.read_csv(\"../input/spaceship-titanic/train.csv\")\n\ndf_test = pd.read_csv(\"../input/spaceship-titanic/test.csv\")\n\ndf_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:47.364331Z","iopub.execute_input":"2022-08-01T16:04:47.364796Z","iopub.status.idle":"2022-08-01T16:04:47.442265Z","shell.execute_reply.started":"2022-08-01T16:04:47.364762Z","shell.execute_reply":"2022-08-01T16:04:47.441096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b>1 <span style='color:lightseagreen'>|</span> EDA </b>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>1.1 | Number of Null</b></p>\n</div>\n<font size=4>Each feature have a few of null, we'll not drop any feature</font>","metadata":{}},{"cell_type":"code","source":"n_null = df_train.isnull().sum()/df_train.shape[0]\n\nfig = px.histogram(x=df_train.columns, y=n_null)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:47.444448Z","iopub.execute_input":"2022-08-01T16:04:47.445324Z","iopub.status.idle":"2022-08-01T16:04:47.516759Z","shell.execute_reply.started":"2022-08-01T16:04:47.445280Z","shell.execute_reply":"2022-08-01T16:04:47.515549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>1.2 | Number of Unique</b></p>\n</div>\n<font size=4>We could find these features have very small unique, such as </font>\n\n\n- HomePlanet\n- CryoSleep\n- Destination\n- Age\n- VIP\n- Transported (this is our predict target)","metadata":{}},{"cell_type":"code","source":"n_unique = df_train.nunique()/df_train.shape[0]\n\nfig = px.histogram(x=df_train.columns, y=n_unique)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:47.518197Z","iopub.execute_input":"2022-08-01T16:04:47.518577Z","iopub.status.idle":"2022-08-01T16:04:47.595506Z","shell.execute_reply.started":"2022-08-01T16:04:47.518544Z","shell.execute_reply":"2022-08-01T16:04:47.594149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>1.3 | Correlation</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"# Let's check the correlation matrix\ndef corr_matrix_plot(df):\n    fig = px.imshow(df.corr(), color_continuous_scale='RdBu_r', origin='lower', text_auto=True, aspect='auto', color_continuous_midpoint=0.0)\n    fig.update_layout(\n        font_size=15\n    )\n    fig.update_layout(title_text=\"Correlation Heatmap\", plot_bgcolor='rgb(242, 242, 242)', paper_bgcolor = 'rgb(242, 242, 242)',\n                      title_font=dict(size=29, family=\"Lato, sans-serif\"), margin=dict(t=90))\n    fig.show()\n\ncorr_matrix_plot(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:47.597699Z","iopub.execute_input":"2022-08-01T16:04:47.598181Z","iopub.status.idle":"2022-08-01T16:04:47.661261Z","shell.execute_reply.started":"2022-08-01T16:04:47.598136Z","shell.execute_reply":"2022-08-01T16:04:47.659346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>1.4 | Categorical Features</b></p>\n</div>\n\n","metadata":{}},{"cell_type":"code","source":"# check the relation of features which have small \"nunique\"\nprint(df_train.nunique())\n\n# HomePlanet\n# 木星、地球、火星\nfig = px.histogram(df_train, x=\"Transported\", color=\"HomePlanet\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs HomePlanet\")\nfig.show()\n\n# CryoSleep\nfig = px.histogram(df_train, x=\"Transported\", color=\"CryoSleep\", barmode=\"group\")\nfig.update_layout(title=\"Transported vs CryoSleep\")\nfig.show()\n\n# Destination\nfig = px.histogram(df_train, x=\"Transported\", color=\"Destination\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Destination\")\nfig.show()\n\n# VIP\n# The count of group plot show : VIP are not a significant feature to Target\n# But, when see the hist percent plot, we get a important infomation!\n# Therefore, dont' drop the \"VIP\" feature now\nfig = px.histogram(df_train, x=\"Transported\", color=\"VIP\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs VIP\")\nfig.show()\n","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-08-01T16:04:47.663025Z","iopub.execute_input":"2022-08-01T16:04:47.663643Z","iopub.status.idle":"2022-08-01T16:04:48.128729Z","shell.execute_reply.started":"2022-08-01T16:04:47.663597Z","shell.execute_reply":"2022-08-01T16:04:48.127028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>1.5 | Numerical Features</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"# Show all the feature dtype\ndf_train.info()\n\n# Get the Numerical feature : 6 features are numerical type\nnumerical_features = df_train.select_dtypes(\"float\").columns\n\n# Create fig layout\nrow, col = 2, 3\nfig = make_subplots(rows=row, cols=col, subplot_titles=numerical_features)\n\ndef grid(*args):\n    return np.stack(np.meshgrid(*args, indexing='ij'), axis=-1)\n\ngrid_layout = grid(np.arange(1, row+1), np.arange(1, col+1)).reshape(row*col, 2)\n\nn = 0\nfor col in numerical_features:\n    fig.add_trace(\n        go.Histogram(\n            x=df_train[col],\n            name=col\n        ),\n        row=grid_layout[n][0], col=grid_layout[n][1]\n    )\n    n += 1\n\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=600,\n)\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:48.130314Z","iopub.execute_input":"2022-08-01T16:04:48.130741Z","iopub.status.idle":"2022-08-01T16:04:48.270918Z","shell.execute_reply.started":"2022-08-01T16:04:48.130708Z","shell.execute_reply":"2022-08-01T16:04:48.269070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b> EDA Conclusion </b><span style='color:orange'>|</span>\n### Correlation:\nRoomService, Spa, VRDeck have some relation with Target\n\n### Categorical features:\n1. VIP == True -> Low Transported True\n2. CryoSleep is important feature, if cryosleep is True the Transported almost be True. (Postive relation)\n3. HomePlanet: Earth - Low Transported True ; Europa - High Transported True\n4. Destination: 55 - High Transported True\n\n### Numerical features:\nMost features value == 0, except \"Age\"\n\nIn missing value step, we can use \"median\"(0) impute most features","metadata":{}},{"cell_type":"markdown","source":"# <b>2 <span style='color:lightseagreen'>|</span> Feature Engineering </b>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>2.1 | Missing Value</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"def imputation(df, features, type_):\n    \"\"\"\n    General imputator through \"mode\" or \"median\"\n    \"\"\"\n    if type_ == \"categorical\":\n        for col in features:\n            # remeber [0] for mode\n            print(f'{col} fill by {df[col].mode()[0]}')\n            df[col] = df[col].fillna(df[col].mode()[0])\n    else:\n        for col in features:\n            print(f'{col} fill by {df[col].median()}')\n            df[col] = df[col].fillna(df[col].median())\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:48.272452Z","iopub.execute_input":"2022-08-01T16:04:48.273270Z","iopub.status.idle":"2022-08-01T16:04:48.281758Z","shell.execute_reply.started":"2022-08-01T16:04:48.273223Z","shell.execute_reply":"2022-08-01T16:04:48.280792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.1.1 | Categorical Features\n\nThe \"Name\" and \"Cabin\" feature need to be handling specially.\n\n* \"Name\" is difficult to guess.\n\n* \"Cabin\" should be transform first.\n\n* \"Cabin\" : X/I/P -> X, I, P (Transform to new features).","metadata":{}},{"cell_type":"code","source":"df_train[\"Age\"]=df_train[\"Age\"].fillna(df_train.groupby('Transported')[\"Age\"].transform('mean'))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:48.283145Z","iopub.execute_input":"2022-08-01T16:04:48.283748Z","iopub.status.idle":"2022-08-01T16:04:48.300253Z","shell.execute_reply.started":"2022-08-01T16:04:48.283713Z","shell.execute_reply":"2022-08-01T16:04:48.298800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imputation the Name feature first\ndf_train[\"Name\"] = df_train[\"Name\"].fillna(\"None\")\ndf_test[\"Name\"] = df_test[\"Name\"].fillna(\"None\")\n\n# Imputation the Cabin feat\ndf_train[\"Cabin\"] = df_train[\"Cabin\"].fillna(method=\"ffill\")\ndf_test[\"Cabin\"] = df_test[\"Cabin\"].fillna(method=\"ffill\")\n\n# Handling the \"Cabin\"\ndef split_Cabin(df):\n    df['Cabin_first'] = df['Cabin'].apply(lambda x: x.split('/')[0])\n    df['Cabin_mid'] = df['Cabin'].apply(lambda x: x.split('/')[1])\n    df['Cabin_last'] = df['Cabin'].apply(lambda x: x.split('/')[2])\n    df.pop('Cabin')\n    return df\n\nsplit_Cabin(df_train)\nsplit_Cabin(df_test)\n\n# Check the new features relation with Target\nfig = px.histogram(df_train, x=\"Transported\", color=\"Cabin_first\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Cabin_first\")\nfig.show()\n\nfig = px.histogram(df_train, x=\"Transported\", color=\"Cabin_mid\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Cabin_mid\")\nfig.show()\n\nfig = px.histogram(df_train, x=\"Transported\", color=\"Cabin_last\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Cabin_last\")\nfig.show()\n\n# As we can see, the cabin_mid is not a good feature\ndf_train.drop(\"Cabin_mid\", axis=1, inplace=True)\ndf_test.drop(\"Cabin_mid\", axis=1, inplace=True)\n\n# Finally, we use \"mode\" to imputation (general)\nnumerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n\nnumerical_features = df_train.select_dtypes(numerics).columns\ncategorical_features = df_train.select_dtypes(exclude=numerics).columns.to_list()\ncategorical_features.remove(\"Transported\")\n\nimputation(df_train, categorical_features, \"categorical\")\nimputation(df_test, categorical_features, \"categorical\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:48.301942Z","iopub.execute_input":"2022-08-01T16:04:48.302312Z","iopub.status.idle":"2022-08-01T16:04:56.537059Z","shell.execute_reply.started":"2022-08-01T16:04:48.302281Z","shell.execute_reply":"2022-08-01T16:04:56.535163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.1.2 | Numerical Features","metadata":{}},{"cell_type":"code","source":"imputation(df_train, numerical_features, \"numerical\")\nimputation(df_test, numerical_features, \"numerical\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:56.538685Z","iopub.execute_input":"2022-08-01T16:04:56.539112Z","iopub.status.idle":"2022-08-01T16:04:56.559938Z","shell.execute_reply.started":"2022-08-01T16:04:56.539069Z","shell.execute_reply":"2022-08-01T16:04:56.559022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>2.2 | Outliers</b></p>\n</div>\n\nRef: https://www.kaggle.com/code/javigallego/space-titanic-exhaustive-eda-fe-optuna\n\nOutliers type:\n\n1. Univariate Outliers Detection\n\n2. Multi-variate Outliers Detection","metadata":{}},{"cell_type":"markdown","source":"### 2.2.1 | Univariate Outliers\n\n* Grubbs Test : detect whether there are outliers in our dataset or not\n* Z-score method : 3 sigma","metadata":{}},{"cell_type":"code","source":"# In our project, we choose Z-score\noutliers = []\n\ndef z_score_detector(df):\n    mean_ = df.mean()\n    std = df.std()\n    \n    for i, value in enumerate(df):\n        z = (value - mean_) / std\n        if abs(z) > 3:\n            outliers.append(i)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:56.567114Z","iopub.execute_input":"2022-08-01T16:04:56.567535Z","iopub.status.idle":"2022-08-01T16:04:56.574349Z","shell.execute_reply.started":"2022-08-01T16:04:56.567499Z","shell.execute_reply":"2022-08-01T16:04:56.573188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.2.2 | Multi-variate Outliers\n\n* DBSCAN\n\nWe are going to analyse outliers in 2 dimensions. (Numerical features with target)","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import DBSCAN\nfrom sklearn.preprocessing import StandardScaler\n\ndef dbscan(df, feature, n):\n    cols = [\"Transported\", feature]\n    X = StandardScaler().fit_transform(df[df[\"Transported\"].isnull() == False][cols].copy().values)\n    \n    db = DBSCAN(eps=3.0, min_samples=10).fit(X)\n    labels = db.labels_ # the cluster label\n\n    # The labels == -1 will called \"outlier\" in DBSCAN\n    outlier_index = np.where(labels ==-1)[0]\n\n    # x-axis : Transported ; y-axis : feature\n    fig.add_trace(\n        go.Scatter(x=X[:, 0], y=X[:, 1],\n                   mode=\"markers\",\n                   marker={\"color\" : labels},\n                   opacity=0.6,\n                   showlegend=False,\n                   ),\n        row=grid_layout[n][0], col=grid_layout[n][1]\n    )\n\n    return outlier_index\n\n# Create fig layout\nrow, col = 2, 3\nfig = make_subplots(rows=row, cols=col, subplot_titles=numerical_features)\n\nfor n, feature in enumerate(numerical_features):\n    # n : number of subplot\n    outliers_index = dbscan(df_train, feature, n)\n    # Drop the outliers\n    df_train.drop(outliers_index, axis=0)\n\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=600,\n    title=\"DBSCAN Outliers\",\n)\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:04:56.576345Z","iopub.execute_input":"2022-08-01T16:04:56.576737Z","iopub.status.idle":"2022-08-01T16:05:03.740653Z","shell.execute_reply.started":"2022-08-01T16:04:56.576703Z","shell.execute_reply":"2022-08-01T16:05:03.738484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>2.3 | Feature Create</b></p>\n</div>\n","metadata":{}},{"cell_type":"markdown","source":"### 2.3.1 | Age\n\nWe could find \"Adult\" will have higher Transported.","metadata":{}},{"cell_type":"code","source":"# Feature Create\ndf_train[\"Adult\"] = (df_train[\"Age\"] >= 18)\ndf_test[\"Adult\"] = (df_test[\"Age\"] >= 18)\n\n# Then check the relation between \"isElder\" and Target\nfig = px.histogram(df_train, x=\"Transported\", color=\"Adult\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Adult\")\nfig.show()\n\n# Drop the Age feature\n# df_train.drop(\"Age\", axis=1, inplace=True)\n# df_test.drop(\"Age\", axis=1, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:03.742477Z","iopub.execute_input":"2022-08-01T16:05:03.742852Z","iopub.status.idle":"2022-08-01T16:05:03.858814Z","shell.execute_reply.started":"2022-08-01T16:05:03.742818Z","shell.execute_reply":"2022-08-01T16:05:03.857464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.3.2 | Bill\n\n<font size=4>The feature about bill:</font>\n\n1. RoomService\n2. FoodCourt\n3. ShoppingMall\n4. Spa\n5. VRDeck\n\n* Let's create \"Spending\" feature through these features\n\n* We could find \"Spending\" more will have lower Transported.\n\n* However, we'll not drop features(RoomService,...) about bill, because there have high correlation with Target.\n","metadata":{}},{"cell_type":"code","source":"total_bill = df_train[[\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\"]].sum(axis=1)\n\n# We could find there are some passengers have no bill (total_bill==0)\nsns.distplot(total_bill)\nplt.show()\n\ndf_train[\"Spending\"] = total_bill\n\n# fig = px.histogram(df_train, x=\"Transported\", color=\"isBill\", barmode=\"group\")\n# fig.update_layout(title=\"Transported vs isBill\")\n# fig.show()\n\n# For testing data\ntotal_bill = df_test[[\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\"]].sum(axis=1)\ndf_test[\"Spending\"] = total_bill","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:03.860348Z","iopub.execute_input":"2022-08-01T16:05:03.861015Z","iopub.status.idle":"2022-08-01T16:05:04.208940Z","shell.execute_reply.started":"2022-08-01T16:05:03.860978Z","shell.execute_reply":"2022-08-01T16:05:04.207786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.2.3 | Qualitative features\n\n1. PassengerId\n2. Name\n3. Cabin (have drop before)\n","metadata":{}},{"cell_type":"code","source":"# PassengerId : xxx_zz\n# We can extract the last number as group\ndf_train[\"id_group\"] = df_train[\"PassengerId\"].str[-1:]\n\nfig = px.histogram(df_train, x=\"Transported\", color=\"id_group\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs idGroup\")\nfig.show()\n\n# Name : firstname_lastname\n# We can extract the lastname as group (family group)\n# However the class is too big, we cannot use these feature easily\ndf_train[\"lastName\"] = df_train[\"Name\"].str.split().str[-1]\n\n# fig = px.histogram(df_train, x=\"Transported\", color=\"lastName\", barmode=\"group\")\n# fig.update_layout(title=\"Transported vs lastName\")\n# fig.show()\n\n# Try another method to make \"lastName\" be a good feature\n# \"Family size\"\ndf_train[\"familySize\"] = df_train.groupby(\"lastName\")[\"PassengerId\"].transform('count')\n\n# But the name == \"None\" which was fill by imputation handling, we should let their familysize == 0\ndf_train.loc[(df_train[\"lastName\"] == \"None\"), \"familySize\"] = 0\n\nfig = px.histogram(df_train, x=\"familySize\", color=\"Transported\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"familySize vs Transported\")\nfig.show()\n\n# For testing data\ndf_test[\"id_group\"] = df_test[\"PassengerId\"].str[-1:]\ndf_test[\"lastName\"] = df_test[\"Name\"].str.split().str[-1]\ndf_test[\"familySize\"] = df_test.groupby(\"lastName\")[\"PassengerId\"].transform('count')\ndf_test.loc[(df_test[\"lastName\"] == \"None\"), \"familySize\"] = 0\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:04.210372Z","iopub.execute_input":"2022-08-01T16:05:04.211324Z","iopub.status.idle":"2022-08-01T16:05:04.468735Z","shell.execute_reply.started":"2022-08-01T16:05:04.211287Z","shell.execute_reply":"2022-08-01T16:05:04.467608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>2.3 | Encoding</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nencode_features = df_train.select_dtypes([\"object\", \"bool\"]).drop([\"PassengerId\", \"Transported\"], axis=1).columns\n\nfor i in encode_features:\n#     Remeber dont encoding \"train\" and \"test\" separately, because\n#     their will cause different result!\n#     df_train[col], _ = df_train[col].factorize()\n#     df_test[col], _ = df_test[col].factorize()\n    le = LabelEncoder()\n    arr = np.concatenate((df_train[i], df_test[i])).astype(str)\n    le.fit(arr)\n    df_train[i] = le.transform(df_train[i].astype(str))\n    df_test[i] = le.transform(df_test[i].astype(str))\n\n    \ndf_train[\"Transported\"].replace([True, False], [1, 0], inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:04.479015Z","iopub.execute_input":"2022-08-01T16:05:04.479415Z","iopub.status.idle":"2022-08-01T16:05:04.675151Z","shell.execute_reply.started":"2022-08-01T16:05:04.479357Z","shell.execute_reply":"2022-08-01T16:05:04.673871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>2.4 | Feature Scaling</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\ndf_train.describe()\n\nscale_features = [\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\", \"Spending\"]\n\nscaler = MinMaxScaler()\ndf_train[scale_features] = scaler.fit_transform(df_train[scale_features])\ndf_test[scale_features] = scaler.transform(df_test[scale_features])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:04.676830Z","iopub.execute_input":"2022-08-01T16:05:04.677188Z","iopub.status.idle":"2022-08-01T16:05:04.745854Z","shell.execute_reply.started":"2022-08-01T16:05:04.677157Z","shell.execute_reply":"2022-08-01T16:05:04.744696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;display:fill;border-radius:8px;\n            background-color:#323232;font-size:150%;\n            font-family:Nexa;letter-spacing:0.5px\">\n    <p style=\"padding: 8px;color:white;\"><b>2.5 | Feature Selection</b></p>\n</div>","metadata":{}},{"cell_type":"code","source":"# Check the correlation\ncorr_matrix_plot(df_train)\n\n# We can find some useless features (relation with transported)\nuseless_features = [\"familySize\", \"Cabin_first\", \"Name\", \"lastName\", \"id_group\", \"ShoppingMall\", \"FoodCourt\", \"VIP\"]\n\ndf_train.drop(useless_features, axis=1, inplace=True)\ndf_test.drop(useless_features, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:04.747372Z","iopub.execute_input":"2022-08-01T16:05:04.748115Z","iopub.status.idle":"2022-08-01T16:05:04.824219Z","shell.execute_reply.started":"2022-08-01T16:05:04.748068Z","shell.execute_reply":"2022-08-01T16:05:04.822978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b>3 <span style='color:lightseagreen'>|</span> Modeling </b>","metadata":{}},{"cell_type":"code","source":"# Modeling\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost.sklearn import XGBClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\n\nfrom sklearn import metrics\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:04.825747Z","iopub.execute_input":"2022-08-01T16:05:04.826777Z","iopub.status.idle":"2022-08-01T16:05:04.833530Z","shell.execute_reply.started":"2022-08-01T16:05:04.826735Z","shell.execute_reply":"2022-08-01T16:05:04.832256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop([\"PassengerId\"], axis=1)\ny = X.pop(\"Transported\")\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.5, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:04.834929Z","iopub.execute_input":"2022-08-01T16:05:04.835284Z","iopub.status.idle":"2022-08-01T16:05:04.855131Z","shell.execute_reply.started":"2022-08-01T16:05:04.835251Z","shell.execute_reply":"2022-08-01T16:05:04.854112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_best_model(X_train, X_test, y_train, y_test):\n    # Logistic Regression\n    logreg = LogisticRegression(max_iter=600, random_state=42)\n    logreg.fit(X_train, y_train)\n    y_pred = logreg.predict(X_test)\n    logreg_acc = round(metrics.accuracy_score(y_test, y_pred), 4) # for the classified accuracy\n    logreg_cv_score = cross_val_score(logreg, X_train, y_train, cv=5).mean()\n\n    # Decision Tree\n    dt = DecisionTreeClassifier(random_state=42)\n    dt.fit(X_train, y_train)\n    y_pred = dt.predict(X_test)\n    dt_acc = round(metrics.accuracy_score(y_test, y_pred), 4)\n    dt_cv_score = cross_val_score(dt, X_train, y_train, cv=5).mean()\n\n    # Random Forest\n    rf = RandomForestClassifier(random_state=42)\n    rf.fit(X_train, y_train)\n    y_pred = rf.predict(X_test)\n    rf_acc = round(metrics.accuracy_score(y_test, y_pred), 4)\n    rf_cv_score = cross_val_score(rf, X_train, y_train, cv=5).mean()\n\n    # XGB\n    xgb = XGBClassifier(random_state=42)\n    xgb.fit(X_train, y_train)\n    y_pred = xgb.predict(X_test)\n    xgb_acc = round(metrics.accuracy_score(y_test, y_pred), 4)\n    xgb_cv_score = cross_val_score(xgb, X_train, y_train, cv=5).mean()\n\n    # Gradient Boosting\n    gb = GradientBoostingClassifier(random_state=42)\n    gb.fit(X_train, y_train)\n    y_pred = gb.predict(X_test)\n    gb_acc = round(metrics.accuracy_score(y_test, y_pred), 4)\n    gb_cv_score = cross_val_score(gb, X_train, y_train, cv=5).mean()\n\n    df_model = pd.DataFrame({\"Model\":[\"LogisticRegression\", \"DecisionTree\", \"RandomForest\", \"XGBoost\", \"GBM\"],\n                           \"Acc\": [logreg_acc, dt_acc, rf_acc, xgb_acc, gb_acc],\n                            \"CvScore\": [logreg_cv_score, dt_cv_score, rf_cv_score, xgb_cv_score, gb_cv_score]})\n    print(df_model)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:04.856863Z","iopub.execute_input":"2022-08-01T16:05:04.857511Z","iopub.status.idle":"2022-08-01T16:05:04.871172Z","shell.execute_reply.started":"2022-08-01T16:05:04.857474Z","shell.execute_reply":"2022-08-01T16:05:04.869453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find_best_model(X_train, X_test, y_train, y_test) I will choose the top3 model for ensembling","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:05:04.872566Z","iopub.execute_input":"2022-08-01T16:05:04.873082Z","iopub.status.idle":"2022-08-01T16:05:04.886492Z","shell.execute_reply.started":"2022-08-01T16:05:04.873049Z","shell.execute_reply":"2022-08-01T16:05:04.885582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b>4 <span style='color:lightseagreen'>|</span> Ensembling </b>\n\n* Soft Voting\n* Ref : https://www.kaggle.com/code/javigallego/space-titanic-exhaustive-eda-fe-optuna","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\n\nlogreg = LogisticRegression()\nxgb = XGBClassifier()\ngbm = GradientBoostingClassifier()\n\nvotingC = VotingClassifier(estimators=[('LogisticReg', logreg), ('XGBoost', xgb), ('GBM', gbm)],\n                           voting='soft',\n                           n_jobs=-1)\n\n# fit training data (df_train)\nvotingC.fit(X,y)\n\nX_test = df_test.drop(\"PassengerId\", axis=1)\n\npred = votingC.predict(X_test)\n\nsub=pd.read_csv('../input/spaceship-titanic/sample_submission.csv')\nsub['Transported'] = pred\nsub['Transported'].replace([0,1], [False, True], inplace=True)\nsub.to_csv('./submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T16:15:01.993883Z","iopub.execute_input":"2022-08-01T16:15:01.994688Z","iopub.status.idle":"2022-08-01T16:15:04.026968Z","shell.execute_reply.started":"2022-08-01T16:15:01.994639Z","shell.execute_reply":"2022-08-01T16:15:04.024916Z"},"trusted":true},"execution_count":null,"outputs":[]}]}