{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T18:35:18.704311Z","iopub.execute_input":"2022-08-07T18:35:18.704619Z","iopub.status.idle":"2022-08-07T18:35:18.710454Z","shell.execute_reply.started":"2022-08-07T18:35:18.704595Z","shell.execute_reply":"2022-08-07T18:35:18.709844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.express as px\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True) # Avoid the notebook cannot show plotly\n\ndf_train = pd.read_csv(\"../input/spaceship-titanic/train.csv\")\n\ndf_test = pd.read_csv(\"../input/spaceship-titanic/test.csv\")\n\ndf_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:18.829143Z","iopub.execute_input":"2022-08-07T18:35:18.829790Z","iopub.status.idle":"2022-08-07T18:35:18.879856Z","shell.execute_reply.started":"2022-08-07T18:35:18.829765Z","shell.execute_reply":"2022-08-07T18:35:18.878621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **1  EDA** \n## **1.1  Number of Null**\n\nEach feature have a few of null, we'll not drop any feature","metadata":{}},{"cell_type":"code","source":"n_null = df_train.isnull().sum()/df_train.shape[0]\n\nfig = px.histogram(x=df_train.columns, y=n_null)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:18.933791Z","iopub.execute_input":"2022-08-07T18:35:18.934648Z","iopub.status.idle":"2022-08-07T18:35:18.984079Z","shell.execute_reply.started":"2022-08-07T18:35:18.934621Z","shell.execute_reply":"2022-08-07T18:35:18.983226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.2  Number of Unique\n\nWe could find these features have very small unique, such as\n* HomePlanet\n* CryoSleep\n* Destination\n* Age\n* VIP\n* Transported (this is our predict target)","metadata":{}},{"cell_type":"code","source":"n_unique = df_train.nunique()/df_train.shape[0]\n\nfig = px.histogram(x=df_train.columns, y=n_unique)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:19.044039Z","iopub.execute_input":"2022-08-07T18:35:19.044946Z","iopub.status.idle":"2022-08-07T18:35:19.102624Z","shell.execute_reply.started":"2022-08-07T18:35:19.044920Z","shell.execute_reply":"2022-08-07T18:35:19.101456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.3 | Correlation","metadata":{}},{"cell_type":"code","source":"# Let's check the correlation matrix\ndef corr_matrix_plot(df):\n    fig = px.imshow(df.corr(), color_continuous_scale='RdBu_r', origin='lower', text_auto=True, aspect='auto', color_continuous_midpoint=0.0)\n    fig.update_layout(\n        font_size=15\n    )\n    fig.update_layout(title_text=\"Correlation Heatmap\", plot_bgcolor='rgb(242, 242, 242)', paper_bgcolor = 'rgb(242, 242, 242)',\n                      title_font=dict(size=29, family=\"Lato, sans-serif\"), margin=dict(t=90))\n    fig.show()\n\ncorr_matrix_plot(df_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:19.145830Z","iopub.execute_input":"2022-08-07T18:35:19.146965Z","iopub.status.idle":"2022-08-07T18:35:19.198820Z","shell.execute_reply.started":"2022-08-07T18:35:19.146922Z","shell.execute_reply":"2022-08-07T18:35:19.198203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.4 | Categorical Features","metadata":{}},{"cell_type":"code","source":"# check the relation of features which have small \"nunique\"\nprint(df_train.nunique())\n\n# HomePlanet\n# 木星、地球、火星\nfig = px.histogram(df_train, x=\"Transported\", color=\"HomePlanet\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs HomePlanet\")\nfig.show()\n\n# CryoSleep\nfig = px.histogram(df_train, x=\"Transported\", color=\"CryoSleep\", barmode=\"group\")\nfig.update_layout(title=\"Transported vs CryoSleep\")\nfig.show()\n\n# Destination\nfig = px.histogram(df_train, x=\"Transported\", color=\"Destination\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Destination\")\nfig.show()\n\n# VIP\n# The count of group plot show : VIP are not a significant feature to Target\n# But, when see the hist percent plot, we get a important infomation!\n# Therefore, dont' drop the \"VIP\" feature now\nfig = px.histogram(df_train, x=\"Transported\", color=\"VIP\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs VIP\")\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:19.243040Z","iopub.execute_input":"2022-08-07T18:35:19.243348Z","iopub.status.idle":"2022-08-07T18:35:19.580269Z","shell.execute_reply.started":"2022-08-07T18:35:19.243322Z","shell.execute_reply":"2022-08-07T18:35:19.579553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.5 | Numerical Features","metadata":{}},{"cell_type":"code","source":"# Show all the feature dtype\ndf_train.info()\n\n# Get the Numerical feature : 6 features are numerical type\nnumerical_features = df_train.select_dtypes(\"float\").columns\n\n# Create fig layout\nrow, col = 2, 3\nfig = make_subplots(rows=row, cols=col, subplot_titles=numerical_features)\n\ndef grid(*args):\n    return np.stack(np.meshgrid(*args, indexing='ij'), axis=-1)\n\ngrid_layout = grid(np.arange(1, row+1), np.arange(1, col+1)).reshape(row*col, 2)\n\nn = 0\nfor col in numerical_features:\n    fig.add_trace(\n        go.Histogram(\n            x=df_train[col],\n            name=col\n        ),\n        row=grid_layout[n][0], col=grid_layout[n][1]\n    )\n    n += 1\n\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=600,\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:19.581509Z","iopub.execute_input":"2022-08-07T18:35:19.582229Z","iopub.status.idle":"2022-08-07T18:35:19.686310Z","shell.execute_reply.started":"2022-08-07T18:35:19.582191Z","shell.execute_reply":"2022-08-07T18:35:19.685683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA Conclusion","metadata":{}},{"cell_type":"markdown","source":"**Correlation:**\n\nRoomService, Spa, VRDeck have some relation with Target\n\n**Categorical features:**\n1.  VIP == True -> Low Transported True\n1.  CryoSleep is important feature, if cryosleep is True the Transported almost be True. (Postive relation)\n1.  HomePlanet: Earth - Low Transported True ; Europa - High Transported True\n1.  Destination: 55 - High Transported True\n\n**Numerical features:**\n\nMost features value == 0, except \"Age\"\n\nIn missing value step, we can use \"median\"(0) impute most features","metadata":{}},{"cell_type":"markdown","source":"# 2 Feature Engineering","metadata":{}},{"cell_type":"markdown","source":" # 2.1 Missing Value","metadata":{}},{"cell_type":"code","source":"def imputation(df, features, type_):\n    \"\"\"\n    General imputator through \"mode\" or \"median\"\n    \"\"\"\n    if type_ == \"categorical\":\n        for col in features:\n            # remeber [0] for mode\n            print(f'{col} fill by {df[col].mode()[0]}')\n            df[col] = df[col].fillna(df[col].mode()[0])\n    else:\n        for col in features:\n            print(f'{col} fill by {df[col].median()}')\n            df[col] = df[col].fillna(df[col].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:19.687873Z","iopub.execute_input":"2022-08-07T18:35:19.688353Z","iopub.status.idle":"2022-08-07T18:35:19.694135Z","shell.execute_reply.started":"2022-08-07T18:35:19.688328Z","shell.execute_reply":"2022-08-07T18:35:19.693025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2.1.1 | Categorical Features**\n\nThe \"Name\" and \"Cabin\" feature need to be handling specially.\n\n* \"Name\" is difficult to guess.\n* \"Cabin\" should be transform first.\n* \"Cabin\" : X/I/P -> X, I, P (Transform to new features).","metadata":{}},{"cell_type":"code","source":"df_train[\"Age\"]=df_train[\"Age\"].fillna(df_train.groupby('Transported')[\"Age\"].transform('mean'))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:19.696836Z","iopub.execute_input":"2022-08-07T18:35:19.697456Z","iopub.status.idle":"2022-08-07T18:35:19.709477Z","shell.execute_reply.started":"2022-08-07T18:35:19.697397Z","shell.execute_reply":"2022-08-07T18:35:19.708575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imputation the Name feature first\ndf_train[\"Name\"] = df_train[\"Name\"].fillna(\"None\")\ndf_test[\"Name\"] = df_test[\"Name\"].fillna(\"None\")\n\n# Imputation the Cabin feat\ndf_train[\"Cabin\"] = df_train[\"Cabin\"].fillna(method=\"ffill\")\ndf_test[\"Cabin\"] = df_test[\"Cabin\"].fillna(method=\"ffill\")\n\n# Handling the \"Cabin\"\ndef split_Cabin(df):\n    df['Cabin_first'] = df['Cabin'].apply(lambda x: x.split('/')[0])\n    df['Cabin_mid'] = df['Cabin'].apply(lambda x: x.split('/')[1])\n    df['Cabin_last'] = df['Cabin'].apply(lambda x: x.split('/')[2])\n    df.pop('Cabin')\n    return df\n\nsplit_Cabin(df_train)\nsplit_Cabin(df_test)\n\n# Check the new features relation with Target\nfig = px.histogram(df_train, x=\"Transported\", color=\"Cabin_first\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Cabin_first\")\nfig.show()\n\nfig = px.histogram(df_train, x=\"Transported\", color=\"Cabin_mid\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Cabin_mid\")\nfig.show()\n\nfig = px.histogram(df_train, x=\"Transported\", color=\"Cabin_last\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Cabin_last\")\nfig.show()\n\n# As we can see, the cabin_mid is not a good feature\ndf_train.drop(\"Cabin_mid\", axis=1, inplace=True)\ndf_test.drop(\"Cabin_mid\", axis=1, inplace=True)\n\n# Finally, we use \"mode\" to imputation (general)\nnumerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n\nnumerical_features = df_train.select_dtypes(numerics).columns\ncategorical_features = df_train.select_dtypes(exclude=numerics).columns.to_list()\ncategorical_features.remove(\"Transported\")\n\nimputation(df_train, categorical_features, \"categorical\")\nimputation(df_test, categorical_features, \"categorical\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:19.711015Z","iopub.execute_input":"2022-08-07T18:35:19.711542Z","iopub.status.idle":"2022-08-07T18:35:25.875390Z","shell.execute_reply.started":"2022-08-07T18:35:19.711518Z","shell.execute_reply":"2022-08-07T18:35:25.874797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.1.2  Numerical Features","metadata":{}},{"cell_type":"code","source":"imputation(df_train, numerical_features, \"numerical\")\nimputation(df_test, numerical_features, \"numerical\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:25.876370Z","iopub.execute_input":"2022-08-07T18:35:25.877221Z","iopub.status.idle":"2022-08-07T18:35:25.891866Z","shell.execute_reply.started":"2022-08-07T18:35:25.877195Z","shell.execute_reply":"2022-08-07T18:35:25.890624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.2 | Outliers\n\n\n**Outliers type:**\n\n* Univariate Outliers Detection\n* \n* Multi-variate Outliers Detection\n\n**2.2.1 | Univariate Outliers**\n* Grubbs Test : detect whether there are outliers in our dataset or not\n* Z-score method : 3 sigma","metadata":{}},{"cell_type":"code","source":"# In our project, we choose Z-score\noutliers = []\n\ndef z_score_detector(df):\n    mean_ = df.mean()\n    std = df.std()\n    \n    for i, value in enumerate(df):\n        z = (value - mean_) / std\n        if abs(z) > 3:\n            outliers.append(i)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:25.893409Z","iopub.execute_input":"2022-08-07T18:35:25.894216Z","iopub.status.idle":"2022-08-07T18:35:25.898904Z","shell.execute_reply.started":"2022-08-07T18:35:25.894174Z","shell.execute_reply":"2022-08-07T18:35:25.898243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import DBSCAN\nfrom sklearn.preprocessing import StandardScaler\n\ndef dbscan(df, feature, n):\n    cols = [\"Transported\", feature]\n    X = StandardScaler().fit_transform(df[df[\"Transported\"].isnull() == False][cols].copy().values)\n    \n    db = DBSCAN(eps=3.0, min_samples=10).fit(X)\n    labels = db.labels_ # the cluster label\n\n    # The labels == -1 will called \"outlier\" in DBSCAN\n    outlier_index = np.where(labels ==-1)[0]\n\n    # x-axis : Transported ; y-axis : feature\n    fig.add_trace(\n        go.Scatter(x=X[:, 0], y=X[:, 1],\n                   mode=\"markers\",\n                   marker={\"color\" : labels},\n                   opacity=0.6,\n                   showlegend=False,\n                   ),\n        row=grid_layout[n][0], col=grid_layout[n][1]\n    )\n    \n    return outlier_index\n\n# Create fig layout\nrow, col = 2, 3\nfig = make_subplots(rows=row, cols=col, subplot_titles=numerical_features)\n\nfor n, feature in enumerate(numerical_features):\n    # n : number of subplot\n    outliers_index = dbscan(df_train, feature, n)\n    # Drop the outliers\n    df_train.drop(outliers_index, axis=0)\n\nfig.update_layout(\n    autosize=False,\n    width=800,\n    height=600,\n    title=\"DBSCAN Outliers\",\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:25.901014Z","iopub.execute_input":"2022-08-07T18:35:25.901976Z","iopub.status.idle":"2022-08-07T18:35:30.230740Z","shell.execute_reply.started":"2022-08-07T18:35:25.901950Z","shell.execute_reply":"2022-08-07T18:35:30.229545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.3 | Feature Create\n\n**2.3.1 | Age**\n\nWe could find \"Adult\" will have higher Transported.","metadata":{}},{"cell_type":"code","source":"# Feature Create\ndf_train[\"Adult\"] = (df_train[\"Age\"] >= 18)\ndf_test[\"Adult\"] = (df_test[\"Age\"] >= 18)\n\n# Then check the relation between \"isElder\" and Target\nfig = px.histogram(df_train, x=\"Transported\", color=\"Adult\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs Adult\")\nfig.show()\n\n# Drop the Age feature\n# df_train.drop(\"Age\", axis=1, inplace=True)\n# df_test.drop(\"Age\", axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:30.232384Z","iopub.execute_input":"2022-08-07T18:35:30.232890Z","iopub.status.idle":"2022-08-07T18:35:30.325217Z","shell.execute_reply.started":"2022-08-07T18:35:30.232864Z","shell.execute_reply":"2022-08-07T18:35:30.324436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **2.3.2 | Bill**\n**The feature about bill:**\n\n1. RoomService\n1. FoodCourt\n1. ShoppingMall\n1. Spa\n1. VRDeck\n\nLet's create \"Spending\" feature through these features\n\nWe could find \"Spending\" more will have lower Transported.\n\nHowever, we'll not drop features(RoomService,...) about bill, because there have high correlation with Target.","metadata":{}},{"cell_type":"code","source":"total_bill = df_train[[\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\"]].sum(axis=1)\n\n# We could find there are some passengers have no bill (total_bill==0)\nsns.distplot(total_bill)\nplt.show()\n\ndf_train[\"Spending\"] = total_bill\n\n# fig = px.histogram(df_train, x=\"Transported\", color=\"isBill\", barmode=\"group\")\n# fig.update_layout(title=\"Transported vs isBill\")\n# fig.show()\n\n# For testing data\ntotal_bill = df_test[[\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\"]].sum(axis=1)\ndf_test[\"Spending\"] = total_bill","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:30.326379Z","iopub.execute_input":"2022-08-07T18:35:30.327570Z","iopub.status.idle":"2022-08-07T18:35:30.812335Z","shell.execute_reply.started":"2022-08-07T18:35:30.327526Z","shell.execute_reply":"2022-08-07T18:35:30.811248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.2.3 | Qualitative features\n\n* PassengerId\n* Name\n* Cabin (have drop before)","metadata":{}},{"cell_type":"code","source":"# PassengerId : xxx_zz\n# We can extract the last number as group\ndf_train[\"id_group\"] = df_train[\"PassengerId\"].str[-1:]\n\nfig = px.histogram(df_train, x=\"Transported\", color=\"id_group\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"Transported vs idGroup\")\nfig.show()\n\n# Name : firstname_lastname\n# We can extract the lastname as group (family group)\n# However the class is too big, we cannot use these feature easily\ndf_train[\"lastName\"] = df_train[\"Name\"].str.split().str[-1]\n\n# fig = px.histogram(df_train, x=\"Transported\", color=\"lastName\", barmode=\"group\")\n# fig.update_layout(title=\"Transported vs lastName\")\n# fig.show()\n\n# Try another method to make \"lastName\" be a good feature\n# \"Family size\"\ndf_train[\"familySize\"] = df_train.groupby(\"lastName\")[\"PassengerId\"].transform('count')\n\n# But the name == \"None\" which was fill by imputation handling, we should let their familysize == 0\ndf_train.loc[(df_train[\"lastName\"] == \"None\"), \"familySize\"] = 0\n\nfig = px.histogram(df_train, x=\"familySize\", color=\"Transported\", barmode=\"group\", histnorm=\"percent\")\nfig.update_layout(title=\"familySize vs Transported\")\nfig.show()\n\n# For testing data\ndf_test[\"id_group\"] = df_test[\"PassengerId\"].str[-1:]\ndf_test[\"lastName\"] = df_test[\"Name\"].str.split().str[-1]\ndf_test[\"familySize\"] = df_test.groupby(\"lastName\")[\"PassengerId\"].transform('count')\ndf_test.loc[(df_test[\"lastName\"] == \"None\"), \"familySize\"] = 0\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:30.813698Z","iopub.execute_input":"2022-08-07T18:35:30.813973Z","iopub.status.idle":"2022-08-07T18:35:31.003378Z","shell.execute_reply.started":"2022-08-07T18:35:30.813948Z","shell.execute_reply":"2022-08-07T18:35:31.002194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.3 | Encoding","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nencode_features = df_train.select_dtypes([\"object\", \"bool\"]).drop([\"PassengerId\", \"Transported\"], axis=1).columns\n\nfor i in encode_features:\n#     Remeber dont encoding \"train\" and \"test\" separately, because\n#     their will cause different result!\n#     df_train[col], _ = df_train[col].factorize()\n#     df_test[col], _ = df_test[col].factorize()\n    le = LabelEncoder()\n    arr = np.concatenate((df_train[i], df_test[i])).astype(str)\n    le.fit(arr)\n    df_train[i] = le.transform(df_train[i].astype(str))\n    df_test[i] = le.transform(df_test[i].astype(str))\n\n    \ndf_train[\"Transported\"].replace([True, False], [1, 0], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:31.004555Z","iopub.execute_input":"2022-08-07T18:35:31.004788Z","iopub.status.idle":"2022-08-07T18:35:31.148869Z","shell.execute_reply.started":"2022-08-07T18:35:31.004766Z","shell.execute_reply":"2022-08-07T18:35:31.147909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\ndf_train.describe()\n\nscale_features = [\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\", \"Spending\"]\n\nscaler = MinMaxScaler()\ndf_train[scale_features] = scaler.fit_transform(df_train[scale_features])\ndf_test[scale_features] = scaler.transform(df_test[scale_features])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:31.150013Z","iopub.execute_input":"2022-08-07T18:35:31.150277Z","iopub.status.idle":"2022-08-07T18:35:31.203585Z","shell.execute_reply.started":"2022-08-07T18:35:31.150253Z","shell.execute_reply":"2022-08-07T18:35:31.202372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Check the correlation\ncorr_matrix_plot(df_train)\n\n# We can find some useless features (relation with transported)\nuseless_features = [\"familySize\", \"Cabin_first\", \"Name\", \"lastName\", \"id_group\", \"ShoppingMall\", \"FoodCourt\", \"VIP\"]\n\ndf_train.drop(useless_features, axis=1, inplace=True)\ndf_test.drop(useless_features, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:31.204940Z","iopub.execute_input":"2022-08-07T18:35:31.205953Z","iopub.status.idle":"2022-08-07T18:35:31.262238Z","shell.execute_reply.started":"2022-08-07T18:35:31.205857Z","shell.execute_reply":"2022-08-07T18:35:31.261055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Modeling\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost.sklearn import XGBClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\n\nfrom sklearn import metrics\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:31.263376Z","iopub.execute_input":"2022-08-07T18:35:31.263910Z","iopub.status.idle":"2022-08-07T18:35:31.270566Z","shell.execute_reply.started":"2022-08-07T18:35:31.263886Z","shell.execute_reply":"2022-08-07T18:35:31.269472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop([\"PassengerId\"], axis=1)\ny = X.pop(\"Transported\")\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.5, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:31.272031Z","iopub.execute_input":"2022-08-07T18:35:31.272315Z","iopub.status.idle":"2022-08-07T18:35:31.285878Z","shell.execute_reply.started":"2022-08-07T18:35:31.272291Z","shell.execute_reply":"2022-08-07T18:35:31.284874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_best_model(X_train, X_test, y_train, y_test):\n    # Logistic Regression\n    logreg = LogisticRegression(max_iter=600, random_state=42)\n    logreg.fit(X_train, y_train)\n    y_pred = logreg.predict(X_test)\n    logreg_acc = round(metrics.accuracy_score(y_test, y_pred), 4) # for the classified accuracy\n    logreg_cv_score = cross_val_score(logreg, X_train, y_train, cv=5).mean()\n\n    # Decision Tree\n    dt = DecisionTreeClassifier(random_state=42)\n    dt.fit(X_train, y_train)\n    y_pred = dt.predict(X_test)\n    dt_acc = round(metrics.accuracy_score(y_test, y_pred), 4)\n    dt_cv_score = cross_val_score(dt, X_train, y_train, cv=5).mean()\n\n    # Random Forest\n    rf = RandomForestClassifier(random_state=42)\n    rf.fit(X_train, y_train)\n    y_pred = rf.predict(X_test)\n    rf_acc = round(metrics.accuracy_score(y_test, y_pred), 4)\n    rf_cv_score = cross_val_score(rf, X_train, y_train, cv=5).mean()\n    # XGB\n    xgb = XGBClassifier(random_state=42)\n    xgb.fit(X_train, y_train)\n    y_pred = xgb.predict(X_test)\n    xgb_acc = round(metrics.accuracy_score(y_test, y_pred), 4)\n    xgb_cv_score = cross_val_score(xgb, X_train, y_train, cv=5).mean()\n\n    # Gradient Boosting\n    gb = GradientBoostingClassifier(random_state=42)\n    gb.fit(X_train, y_train)\n    y_pred = gb.predict(X_test)\n    gb_acc = round(metrics.accuracy_score(y_test, y_pred), 4)\n    gb_cv_score = cross_val_score(gb, X_train, y_train, cv=5).mean()\n\n    df_model = pd.DataFrame({\"Model\":[\"LogisticRegression\", \"DecisionTree\", \"RandomForest\", \"XGBoost\", \"GBM\"],\n                           \"Acc\": [logreg_acc, dt_acc, rf_acc, xgb_acc, gb_acc],\n                            \"CvScore\": [logreg_cv_score, dt_cv_score, rf_cv_score, xgb_cv_score, gb_cv_score]})\n    print(df_model)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:31.287384Z","iopub.execute_input":"2022-08-07T18:35:31.287746Z","iopub.status.idle":"2022-08-07T18:35:31.299835Z","shell.execute_reply.started":"2022-08-07T18:35:31.287719Z","shell.execute_reply":"2022-08-07T18:35:31.298774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\n\nlogreg = LogisticRegression()\nxgb = XGBClassifier()\ngbm = GradientBoostingClassifier()\n\nvotingC = VotingClassifier(estimators=[('LogisticReg', logreg), ('XGBoost', xgb), ('GBM', gbm)],\n                           voting='soft',\n                           n_jobs=-1)\n\n# fit training data (df_train)\nvotingC.fit(X,y)\n\nX_test = df_test.drop(\"PassengerId\", axis=1)\n\npred = votingC.predict(X_test)\n\nsub=pd.read_csv('../input/spaceship-titanic/sample_submission.csv')\nsub['Transported'] = pred\nsub['Transported'].replace([0,1], [False, True], inplace=True)\nsub.to_csv('./submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T18:35:31.303295Z","iopub.execute_input":"2022-08-07T18:35:31.303843Z","iopub.status.idle":"2022-08-07T18:35:33.869665Z","shell.execute_reply.started":"2022-08-07T18:35:31.303817Z","shell.execute_reply":"2022-08-07T18:35:33.868757Z"},"trusted":true},"execution_count":null,"outputs":[]}]}