{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:32.534937Z","iopub.execute_input":"2022-08-05T18:45:32.535555Z","iopub.status.idle":"2022-08-05T18:45:32.572533Z","shell.execute_reply.started":"2022-08-05T18:45:32.535421Z","shell.execute_reply":"2022-08-05T18:45:32.571213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import (OrdinalEncoder, StandardScaler,\n                                   OneHotEncoder)\nfrom sklearn.model_selection import cross_val_score, train_test_split\n\nfrom sklearn.feature_selection import mutual_info_regression\nfrom sklearn.impute import KNNImputer\n\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import RepeatedStratifiedKFold\n\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:32.606423Z","iopub.execute_input":"2022-08-05T18:45:32.607573Z","iopub.status.idle":"2022-08-05T18:45:34.056945Z","shell.execute_reply.started":"2022-08-05T18:45:32.607522Z","shell.execute_reply":"2022-08-05T18:45:34.055581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" #### Init basic functions for data analysis","metadata":{}},{"cell_type":"code","source":"# Create data description function\ndef data_desc(df):\n    print()\n    print(\"Overall Data Description\")\n    print(\"Total number of records\", df.shape[0])\n    print(\"Total number of columns/features\", df.shape[1])\n    print(\"\")\n    cols = df.columns\n    data_type = []\n    for col in df.columns:\n        data_type.append(df[col].dtype)\n    n_uni = df.nunique()\n    n_miss = df.isna().sum()\n    names = list(zip(cols, data_type, n_uni, n_miss))\n    variable_desc = pd.DataFrame(\n        names, columns=[\"Name\", \"Type\", \"Unique levels\", \"Missing\"])\n    print(variable_desc)\n\n\n# Function to calculate the mutual information\ndef create_mi_score(X, y):\n    mi_scores = mutual_info_regression(X, y)\n    mi_scores = pd.DataFrame(mi_scores, columns=['mi_score'], index=X.columns)\n    mi_scores = mi_scores.sort_values(by='mi_score', ascending=False)\n    return mi_scores\n\n\n# Plot the mutual information score\ndef plot_mi_scores(scores):\n    plt.figure(dpi=100, figsize=(8, 5))\n    scores = scores.sort_values(by='mi_score', ascending=True)\n    width = np.arange(len(scores))\n    ticks = list(scores.index)\n    plt.barh(width, scores['mi_score'])\n    plt.yticks(width, ticks)\n    plt.title(\"Mutual Information Scores\")\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.059290Z","iopub.execute_input":"2022-08-05T18:45:34.060499Z","iopub.status.idle":"2022-08-05T18:45:34.075997Z","shell.execute_reply.started":"2022-08-05T18:45:34.060448Z","shell.execute_reply":"2022-08-05T18:45:34.074667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" Read in the data and first glance at the data","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('../input/spaceship-titanic/train.csv')\ntest_data = pd.read_csv('../input/spaceship-titanic/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.078493Z","iopub.execute_input":"2022-08-05T18:45:34.078819Z","iopub.status.idle":"2022-08-05T18:45:34.165209Z","shell.execute_reply.started":"2022-08-05T18:45:34.078790Z","shell.execute_reply":"2022-08-05T18:45:34.164208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add field that mark the dataset as train or test\ntrain_data['dataset'] = 'train'\ntest_data['dataset'] = 'test'\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.168353Z","iopub.execute_input":"2022-08-05T18:45:34.169113Z","iopub.status.idle":"2022-08-05T18:45:34.181544Z","shell.execute_reply.started":"2022-08-05T18:45:34.169064Z","shell.execute_reply":"2022-08-05T18:45:34.180680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Describe the data\ntrain_data.describe()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.183412Z","iopub.execute_input":"2022-08-05T18:45:34.184240Z","iopub.status.idle":"2022-08-05T18:45:34.244365Z","shell.execute_reply.started":"2022-08-05T18:45:34.184179Z","shell.execute_reply":"2022-08-05T18:45:34.243066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the first 5 rows of the data\ntrain_data.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.246015Z","iopub.execute_input":"2022-08-05T18:45:34.246973Z","iopub.status.idle":"2022-08-05T18:45:34.271202Z","shell.execute_reply.started":"2022-08-05T18:45:34.246935Z","shell.execute_reply":"2022-08-05T18:45:34.269976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df compose of the train and the test\ndf = pd.concat([train_data, test_data], axis=0)\ndf.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.272581Z","iopub.execute_input":"2022-08-05T18:45:34.273749Z","iopub.status.idle":"2022-08-05T18:45:34.288521Z","shell.execute_reply.started":"2022-08-05T18:45:34.273702Z","shell.execute_reply":"2022-08-05T18:45:34.287313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.289638Z","iopub.execute_input":"2022-08-05T18:45:34.290345Z","iopub.status.idle":"2022-08-05T18:45:34.314244Z","shell.execute_reply.started":"2022-08-05T18:45:34.290310Z","shell.execute_reply":"2022-08-05T18:45:34.313082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_desc(df)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.316295Z","iopub.execute_input":"2022-08-05T18:45:34.317058Z","iopub.status.idle":"2022-08-05T18:45:34.357391Z","shell.execute_reply.started":"2022-08-05T18:45:34.317013Z","shell.execute_reply":"2022-08-05T18:45:34.356085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the data in df\ndf.hist(bins=12, figsize=(20, 15))\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:34.362242Z","iopub.execute_input":"2022-08-05T18:45:34.363422Z","iopub.status.idle":"2022-08-05T18:45:35.345256Z","shell.execute_reply.started":"2022-08-05T18:45:34.363370Z","shell.execute_reply":"2022-08-05T18:45:35.344327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_extractor(df):\n    result = df['PassengerId'].str.split('_', expand=True).astype(float)\n    result.rename(columns={\n        0: 'PassengerGroup',\n        1: 'PassengerNumber'\n    },\n                  inplace=True)\n\n    cabin_cols = pd.DataFrame(df.Cabin.str.split('/', expand=True))\n    result = pd.concat([result, cabin_cols], axis=1)\n    result.rename(columns={0: 'Deck', 1: 'Num', 2: 'Side'}, inplace=True)\n    result['Num'] = result['Num'].astype(float)\n\n    result = pd.concat([\n        result, df[[\n            'HomePlanet',\n            'CryoSleep',\n            'Destination',\n            'Age',\n            'VIP',\n            'RoomService',\n            'FoodCourt',\n            'ShoppingMall',\n            'Spa',\n            'VRDeck',\n        ]]\n    ],\n                       axis=1)\n\n    # Split name to first and last name\n    result[['Name', 'Surname']] = df['Name'].str.split(' ', expand=True)\n\n    return result\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.346811Z","iopub.execute_input":"2022-08-05T18:45:35.347889Z","iopub.status.idle":"2022-08-05T18:45:35.360060Z","shell.execute_reply.started":"2022-08-05T18:45:35.347839Z","shell.execute_reply":"2022-08-05T18:45:35.358554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the features to test and train data\nX_train_full = feature_extractor(train_data)\ny_train_full = train_data['Transported']\n\nX_valid = feature_extractor(test_data)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.361801Z","iopub.execute_input":"2022-08-05T18:45:35.362844Z","iopub.status.idle":"2022-08-05T18:45:35.603599Z","shell.execute_reply.started":"2022-08-05T18:45:35.362799Z","shell.execute_reply":"2022-08-05T18:45:35.602306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" #### Check Name and Surname, probably we can use it","metadata":{}},{"cell_type":"code","source":"X_train_full['Name'].value_counts().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.605181Z","iopub.execute_input":"2022-08-05T18:45:35.605662Z","iopub.status.idle":"2022-08-05T18:45:35.616798Z","shell.execute_reply.started":"2022-08-05T18:45:35.605606Z","shell.execute_reply":"2022-08-05T18:45:35.615587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_full['Surname'].value_counts().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.618687Z","iopub.execute_input":"2022-08-05T18:45:35.619441Z","iopub.status.idle":"2022-08-05T18:45:35.634457Z","shell.execute_reply.started":"2022-08-05T18:45:35.619384Z","shell.execute_reply":"2022-08-05T18:45:35.633291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_full.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.635947Z","iopub.execute_input":"2022-08-05T18:45:35.637078Z","iopub.status.idle":"2022-08-05T18:45:35.672065Z","shell.execute_reply.started":"2022-08-05T18:45:35.637032Z","shell.execute_reply":"2022-08-05T18:45:35.670744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_encoder(df):\n    columns_to_encode = [\n        'HomePlanet', 'CryoSleep', 'Deck', 'Side', 'Destination', 'VIP',\n        'Name', 'Surname'\n    ]\n    transformer = OrdinalEncoder()\n    transformer.fit(df[columns_to_encode])\n    df[columns_to_encode] = transformer.transform(df[columns_to_encode])\n    return df\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.673851Z","iopub.execute_input":"2022-08-05T18:45:35.674512Z","iopub.status.idle":"2022-08-05T18:45:35.682441Z","shell.execute_reply.started":"2022-08-05T18:45:35.674466Z","shell.execute_reply":"2022-08-05T18:45:35.681103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode the features\nX_train_full = feature_encoder(X_train_full)\nX_valid = feature_encoder(X_valid)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.684494Z","iopub.execute_input":"2022-08-05T18:45:35.685329Z","iopub.status.idle":"2022-08-05T18:45:35.786953Z","shell.execute_reply.started":"2022-08-05T18:45:35.685284Z","shell.execute_reply":"2022-08-05T18:45:35.786041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_full.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.788847Z","iopub.execute_input":"2022-08-05T18:45:35.789711Z","iopub.status.idle":"2022-08-05T18:45:35.817899Z","shell.execute_reply.started":"2022-08-05T18:45:35.789662Z","shell.execute_reply":"2022-08-05T18:45:35.816706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_full.describe()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.819800Z","iopub.execute_input":"2022-08-05T18:45:35.820128Z","iopub.status.idle":"2022-08-05T18:45:35.893380Z","shell.execute_reply.started":"2022-08-05T18:45:35.820093Z","shell.execute_reply":"2022-08-05T18:45:35.892001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_valid.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.895414Z","iopub.execute_input":"2022-08-05T18:45:35.896331Z","iopub.status.idle":"2022-08-05T18:45:35.926562Z","shell.execute_reply.started":"2022-08-05T18:45:35.896269Z","shell.execute_reply":"2022-08-05T18:45:35.925327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop nan values and run mutual info regression\nX_train_full_dropped_na = pd.concat([X_train_full, y_train_full],\n                                    axis=1).dropna()\n\ny_train_full_dropped_na = X_train_full_dropped_na.pop('Transported')\n\nX_train_full_dropped_na.shape, y_train_full_dropped_na.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.928503Z","iopub.execute_input":"2022-08-05T18:45:35.929424Z","iopub.status.idle":"2022-08-05T18:45:35.948820Z","shell.execute_reply.started":"2022-08-05T18:45:35.929374Z","shell.execute_reply":"2022-08-05T18:45:35.947314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the mutual information score and plot it\nmi_scores = create_mi_score(X_train_full_dropped_na, y_train_full_dropped_na)\n\nplot_mi_scores(mi_scores)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:35.950819Z","iopub.execute_input":"2022-08-05T18:45:35.951310Z","iopub.status.idle":"2022-08-05T18:45:36.922625Z","shell.execute_reply.started":"2022-08-05T18:45:35.951264Z","shell.execute_reply":"2022-08-05T18:45:36.921312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_na_values_proportion(df, title):\n    plt.figure(figsize=(8, 4))\n    sns.displot(data=df.isna().melt(value_name=\"missing\"),\n                y=\"variable\",\n                hue=\"missing\",\n                multiple=\"fill\",\n                aspect=2)\n    plt.title(title)\n    plt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:36.924409Z","iopub.execute_input":"2022-08-05T18:45:36.925148Z","iopub.status.idle":"2022-08-05T18:45:36.932785Z","shell.execute_reply.started":"2022-08-05T18:45:36.925101Z","shell.execute_reply":"2022-08-05T18:45:36.931512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the missing value proportion for train\nplot_na_values_proportion(X_train_full,\n                          \"Missing Value Proportion Each Feature: Train Data\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:36.934421Z","iopub.execute_input":"2022-08-05T18:45:36.934888Z","iopub.status.idle":"2022-08-05T18:45:37.734560Z","shell.execute_reply.started":"2022-08-05T18:45:36.934845Z","shell.execute_reply":"2022-08-05T18:45:37.733300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the missing value proportion for valid\nplot_na_values_proportion(X_valid,\n                          \"Missing Value Proportion Each Feature: Valid Data\")\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:37.736639Z","iopub.execute_input":"2022-08-05T18:45:37.737098Z","iopub.status.idle":"2022-08-05T18:45:38.442583Z","shell.execute_reply.started":"2022-08-05T18:45:37.737055Z","shell.execute_reply":"2022-08-05T18:45:38.441449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create several features from the data\ndef add_features(df):\n    df['ServicesSpend'] = df['RoomService'] + df['Spa'] + df['VRDeck']\n    df['FoodSpend'] = df['FoodCourt'] + df['ShoppingMall']\n    df['AllSpend'] = df['ServicesSpend'] + df['FoodSpend']\n    df['SpendRatio'] = df['FoodSpend'] / (df['ServicesSpend'] +\n                                          np.finfo(float).eps)\n    return df\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:38.444130Z","iopub.execute_input":"2022-08-05T18:45:38.445332Z","iopub.status.idle":"2022-08-05T18:45:38.453123Z","shell.execute_reply.started":"2022-08-05T18:45:38.445283Z","shell.execute_reply":"2022-08-05T18:45:38.451668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def imputer_knn(df, n_neighbors=5):\n    imputer = KNNImputer(n_neighbors=n_neighbors)\n    result = imputer.fit_transform(df)\n    result = pd.DataFrame(result, columns=df.columns, index=df.index)\n    return result\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:38.454324Z","iopub.execute_input":"2022-08-05T18:45:38.454677Z","iopub.status.idle":"2022-08-05T18:45:38.464763Z","shell.execute_reply.started":"2022-08-05T18:45:38.454646Z","shell.execute_reply":"2022-08-05T18:45:38.463688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Impute the missing values\nX_train_final_knn = imputer_knn(X_train_full, 2)\n\nX_valid_final_knn = imputer_knn(X_valid, 2)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:38.472505Z","iopub.execute_input":"2022-08-05T18:45:38.473451Z","iopub.status.idle":"2022-08-05T18:45:40.459829Z","shell.execute_reply.started":"2022-08-05T18:45:38.473404Z","shell.execute_reply":"2022-08-05T18:45:40.458614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the mutual information score from imputed data and plot it\nmi_scores_imputed = create_mi_score(X_train_final_knn, y_train_full)\n\nplot_mi_scores(mi_scores_imputed)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:40.461667Z","iopub.execute_input":"2022-08-05T18:45:40.462162Z","iopub.status.idle":"2022-08-05T18:45:41.678248Z","shell.execute_reply.started":"2022-08-05T18:45:40.462115Z","shell.execute_reply":"2022-08-05T18:45:41.677422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_final_knn = add_features(X_train_final_knn)\nX_valid_final_knn = add_features(X_valid_final_knn)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:41.679875Z","iopub.execute_input":"2022-08-05T18:45:41.681031Z","iopub.status.idle":"2022-08-05T18:45:41.694756Z","shell.execute_reply.started":"2022-08-05T18:45:41.680984Z","shell.execute_reply":"2022-08-05T18:45:41.693340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the mutual information with new features\nmi_scores_new_features = create_mi_score(X_train_final_knn, y_train_full)\n\nplot_mi_scores(mi_scores_new_features)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:41.697632Z","iopub.execute_input":"2022-08-05T18:45:41.699333Z","iopub.status.idle":"2022-08-05T18:45:43.179285Z","shell.execute_reply.started":"2022-08-05T18:45:41.699282Z","shell.execute_reply":"2022-08-05T18:45:43.178310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" #### I see any changes from imputing the missing values by knn, probably not the best way...","metadata":{}},{"cell_type":"markdown","source":" ### Create a column transformer to scale the data\n and encode the categorical features","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nonehot = OneHotEncoder(categories='auto')\n\ncolumn_transformer = ColumnTransformer([\n    ('standardize', scaler, [\n        'PassengerGroup',\n        'Deck',\n        'Num',\n        'Age',\n        'Name',\n        'Surname',\n        'RoomService',\n        'FoodCourt',\n        'ShoppingMall',\n        'Spa',\n        'VRDeck',\n        'ServicesSpend',\n        'FoodSpend',\n        'AllSpend',\n        'SpendRatio',\n    ]),\n    ('one_hot', onehot, [\n        'PassengerNumber',\n        'Side',\n        'HomePlanet',\n        'CryoSleep',\n        'Destination',\n        'VIP',\n    ])\n])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:43.180834Z","iopub.execute_input":"2022-08-05T18:45:43.181729Z","iopub.status.idle":"2022-08-05T18:45:43.188861Z","shell.execute_reply.started":"2022-08-05T18:45:43.181679Z","shell.execute_reply":"2022-08-05T18:45:43.187716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ### Train the model XGBRegressor and knn imputed data","metadata":{}},{"cell_type":"code","source":"# Create the model and pipe for imputed data\n\n\ndef create_model_xgb(column_transformer, X, y):\n    model = XGBClassifier()\n    pipe_xgb = make_pipeline(column_transformer, model)\n    scores = cross_val_score(pipe_xgb,\n                             X,\n                             y.astype(int),\n                             scoring=\"accuracy\",\n                             cv=5,\n                             verbose=1)\n    return scores, pipe_xgb\n\n\ndef create_model_catboost(column_transformer, X, y):\n    model = CatBoostClassifier(eval_metric='Accuracy',\n                               use_best_model=True,\n                               random_seed=42,\n                               iterations=100)\n    X_transformed = column_transformer.fit_transform(X)\n    X_train, X_val, y_train, y_val = train_test_split(X_transformed,\n                                                      y.astype(int),\n                                                      test_size=0.2,\n                                                      random_state=42)\n    model.fit(X_train, y_train, eval_set=(X_val, y_val), plot=True)\n    return model\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:43.190836Z","iopub.execute_input":"2022-08-05T18:45:43.191680Z","iopub.status.idle":"2022-08-05T18:45:43.202168Z","shell.execute_reply.started":"2022-08-05T18:45:43.191636Z","shell.execute_reply":"2022-08-05T18:45:43.201107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores, pipe_xgb = create_model_xgb(column_transformer, X_train_final_knn,\n                                    y_train_full)\n\nscores\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:43.203791Z","iopub.execute_input":"2022-08-05T18:45:43.204388Z","iopub.status.idle":"2022-08-05T18:45:51.384361Z","shell.execute_reply.started":"2022-08-05T18:45:43.204345Z","shell.execute_reply":"2022-08-05T18:45:51.383354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores.mean()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:51.388170Z","iopub.execute_input":"2022-08-05T18:45:51.390695Z","iopub.status.idle":"2022-08-05T18:45:51.399366Z","shell.execute_reply.started":"2022-08-05T18:45:51.390651Z","shell.execute_reply":"2022-08-05T18:45:51.398286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" #### Score looks bad for XGBRegressor with knn imputed data","metadata":{}},{"cell_type":"code","source":"catboost_model = create_model_catboost(column_transformer, X_train_final_knn,\n                                       y_train_full)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:51.401004Z","iopub.execute_input":"2022-08-05T18:45:51.401655Z","iopub.status.idle":"2022-08-05T18:45:52.330360Z","shell.execute_reply.started":"2022-08-05T18:45:51.401622Z","shell.execute_reply":"2022-08-05T18:45:52.329033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" #### With catboost, the score is much better, but looks overfitted","metadata":{}},{"cell_type":"markdown","source":" ### Let's use different imputation methods","metadata":{}},{"cell_type":"code","source":"\n\ndef get_random_field_val(df, field):\n    return df[field][df[field].notna()].sample(1).values[0]\n\n\ndef impute_data(data_frame):\n    df = data_frame.copy()\n    df.fillna(method='bfill', axis=0)\n    df.fillna(method='bfill', axis=0)\n    df = df.fillna({\n        'HomePlanet': get_random_field_val(df, 'HomePlanet'),\n        'CryoSleep': get_random_field_val(df, 'CryoSleep'),\n        'Deck': get_random_field_val(df, 'Deck'),\n        'Num': get_random_field_val(df, 'Num'),\n        'Side': get_random_field_val(df, 'Side'),\n        'Destination': get_random_field_val(df, 'Destination'),\n        'Age': df.Age.median(),\n        'VIP': get_random_field_val(df, 'VIP'),\n        'RoomService': df.RoomService.median(),\n        'FoodCourt': df.FoodCourt.median(),\n        'ShoppingMall': df.ShoppingMall.median(),\n        'Spa': df.Spa.median(),\n        'VRDeck': df.VRDeck.median(),\n        'Name': get_random_field_val(df, 'Name'),\n        'Surname': get_random_field_val(df, 'Surname'),\n    })\n    return df\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:52.332258Z","iopub.execute_input":"2022-08-05T18:45:52.333444Z","iopub.status.idle":"2022-08-05T18:45:52.344147Z","shell.execute_reply.started":"2022-08-05T18:45:52.333396Z","shell.execute_reply":"2022-08-05T18:45:52.342762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_final_custom_imputation = add_features(impute_data(X_train_full))\n\nX_valid_final_custom_imputation = add_features(impute_data(X_valid))\n\nX_train_final_custom_imputation\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:52.345453Z","iopub.execute_input":"2022-08-05T18:45:52.345783Z","iopub.status.idle":"2022-08-05T18:45:52.463605Z","shell.execute_reply.started":"2022-08-05T18:45:52.345754Z","shell.execute_reply":"2022-08-05T18:45:52.462332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores, pipe_xgb = create_model_xgb(column_transformer,\n                                    X_train_final_custom_imputation,\n                                    y_train_full)\n\nscores\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:52.465477Z","iopub.execute_input":"2022-08-05T18:45:52.465997Z","iopub.status.idle":"2022-08-05T18:45:58.917887Z","shell.execute_reply.started":"2022-08-05T18:45:52.465950Z","shell.execute_reply":"2022-08-05T18:45:58.916868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores.mean()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:58.919478Z","iopub.execute_input":"2022-08-05T18:45:58.920150Z","iopub.status.idle":"2022-08-05T18:45:58.927254Z","shell.execute_reply.started":"2022-08-05T18:45:58.920114Z","shell.execute_reply":"2022-08-05T18:45:58.926085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catboost_model = create_model_catboost(column_transformer,\n                                       X_train_final_custom_imputation,\n                                       y_train_full)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:58.930403Z","iopub.execute_input":"2022-08-05T18:45:58.931123Z","iopub.status.idle":"2022-08-05T18:45:59.699419Z","shell.execute_reply.started":"2022-08-05T18:45:58.931085Z","shell.execute_reply":"2022-08-05T18:45:59.698277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_transformed = column_transformer.fit_transform(X_valid_final_custom_imputation)\npred = catboost_model.predict(X_transformed)\npred = pred.astype(bool)\n\nsubmission = pd.DataFrame({'Transported': pred}, index=test_data.PassengerId)\nsubmission.to_csv('submission_catboost_model.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:59.701278Z","iopub.execute_input":"2022-08-05T18:45:59.702128Z","iopub.status.idle":"2022-08-05T18:45:59.789478Z","shell.execute_reply.started":"2022-08-05T18:45:59.702092Z","shell.execute_reply":"2022-08-05T18:45:59.788514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" #### With catboost, the score is much better, but looks overfitted.\n #### I reached #155 in the leaderboard, score ~0.808 but I think I should try to improve the model\n ### Check the ensemble of models","metadata":{}},{"cell_type":"code","source":"# Create the VotingClassifier\nvoting_clf = VotingClassifier(\n    estimators=[\n        ('KNN', KNeighborsClassifier()),\n        ('DTree', DecisionTreeClassifier()),\n        ('LogReg', LogisticRegression(solver='liblinear',\n                                    multi_class='auto',\n                                    max_iter=1000)),\n        ('XGB', XGBClassifier()),\n    ],\n    voting='hard')\n\npipe_vc = make_pipeline(column_transformer, voting_clf)\npipe_vc.fit(X_train_final_custom_imputation, y_train_full)\n\ncv_rskf = RepeatedStratifiedKFold(n_splits=10, n_repeats=3, random_state=1)\n\nscores = cross_val_score(\n    pipe_vc,\n    X_train_final_custom_imputation,\n    y_train_full.astype(int),\n    scoring='accuracy',\n    cv=cv_rskf,\n    n_jobs=-1,\n    error_score='raise'\n)\n\nscores\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:45:59.791102Z","iopub.execute_input":"2022-08-05T18:45:59.791478Z","iopub.status.idle":"2022-08-05T18:46:40.011664Z","shell.execute_reply.started":"2022-08-05T18:45:59.791447Z","shell.execute_reply":"2022-08-05T18:46:40.010610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores.mean()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:46:40.013421Z","iopub.execute_input":"2022-08-05T18:46:40.014617Z","iopub.status.idle":"2022-08-05T18:46:40.021847Z","shell.execute_reply.started":"2022-08-05T18:46:40.014577Z","shell.execute_reply":"2022-08-05T18:46:40.020905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = pipe_vc.predict(X_valid_final_custom_imputation)\npred = pred.astype(bool)\n\nsubmission = pd.DataFrame({'Transported': pred}, index=test_data.PassengerId)\nsubmission.to_csv('submission_ensemble_model.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T18:48:43.465910Z","iopub.execute_input":"2022-08-05T18:48:43.466401Z","iopub.status.idle":"2022-08-05T18:48:44.579649Z","shell.execute_reply.started":"2022-08-05T18:48:43.466362Z","shell.execute_reply":"2022-08-05T18:48:44.578377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}