{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !kaggle competitions download -c spaceship-titanic","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.197844Z","iopub.execute_input":"2022-07-28T13:05:41.198280Z","iopub.status.idle":"2022-07-28T13:05:41.203777Z","shell.execute_reply.started":"2022-07-28T13:05:41.198245Z","shell.execute_reply":"2022-07-28T13:05:41.202334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from zipfile import ZipFile","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.228447Z","iopub.execute_input":"2022-07-28T13:05:41.229249Z","iopub.status.idle":"2022-07-28T13:05:41.233137Z","shell.execute_reply.started":"2022-07-28T13:05:41.229197Z","shell.execute_reply":"2022-07-28T13:05:41.232313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with ZipFile('./spaceship-titanic.zip', 'r') as zipObj:\n#     zipObj.extractall()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.246687Z","iopub.execute_input":"2022-07-28T13:05:41.247090Z","iopub.status.idle":"2022-07-28T13:05:41.251678Z","shell.execute_reply.started":"2022-07-28T13:05:41.247057Z","shell.execute_reply":"2022-07-28T13:05:41.250498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.267426Z","iopub.execute_input":"2022-07-28T13:05:41.268016Z","iopub.status.idle":"2022-07-28T13:05:41.272532Z","shell.execute_reply.started":"2022-07-28T13:05:41.267982Z","shell.execute_reply":"2022-07-28T13:05:41.271526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.282611Z","iopub.execute_input":"2022-07-28T13:05:41.283287Z","iopub.status.idle":"2022-07-28T13:05:41.288659Z","shell.execute_reply.started":"2022-07-28T13:05:41.283252Z","shell.execute_reply":"2022-07-28T13:05:41.287651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/spaceship-titanic/train.csv')\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.299208Z","iopub.execute_input":"2022-07-28T13:05:41.299907Z","iopub.status.idle":"2022-07-28T13:05:41.395787Z","shell.execute_reply.started":"2022-07-28T13:05:41.299860Z","shell.execute_reply":"2022-07-28T13:05:41.394982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['CryoSleep'] = train_df['CryoSleep'].astype(bool)\ntrain_df['VIP'] = train_df['VIP'].astype(bool)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.397186Z","iopub.execute_input":"2022-07-28T13:05:41.397811Z","iopub.status.idle":"2022-07-28T13:05:41.409722Z","shell.execute_reply.started":"2022-07-28T13:05:41.397775Z","shell.execute_reply":"2022-07-28T13:05:41.408813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.410793Z","iopub.execute_input":"2022-07-28T13:05:41.412018Z","iopub.status.idle":"2022-07-28T13:05:41.448135Z","shell.execute_reply.started":"2022-07-28T13:05:41.411968Z","shell.execute_reply":"2022-07-28T13:05:41.446951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.451412Z","iopub.execute_input":"2022-07-28T13:05:41.452219Z","iopub.status.idle":"2022-07-28T13:05:41.492269Z","shell.execute_reply.started":"2022-07-28T13:05:41.452171Z","shell.execute_reply":"2022-07-28T13:05:41.490840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"sns.countplot(data=train_df, x='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.493940Z","iopub.execute_input":"2022-07-28T13:05:41.494441Z","iopub.status.idle":"2022-07-28T13:05:41.663831Z","shell.execute_reply.started":"2022-07-28T13:05:41.494394Z","shell.execute_reply":"2022-07-28T13:05:41.662377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='HomePlanet', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.665679Z","iopub.execute_input":"2022-07-28T13:05:41.666994Z","iopub.status.idle":"2022-07-28T13:05:41.904991Z","shell.execute_reply.started":"2022-07-28T13:05:41.666933Z","shell.execute_reply":"2022-07-28T13:05:41.903845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='CryoSleep', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:41.906298Z","iopub.execute_input":"2022-07-28T13:05:41.906615Z","iopub.status.idle":"2022-07-28T13:05:42.128825Z","shell.execute_reply.started":"2022-07-28T13:05:41.906587Z","shell.execute_reply":"2022-07-28T13:05:42.127508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='Destination', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:42.130720Z","iopub.execute_input":"2022-07-28T13:05:42.131457Z","iopub.status.idle":"2022-07-28T13:05:42.365463Z","shell.execute_reply.started":"2022-07-28T13:05:42.131417Z","shell.execute_reply":"2022-07-28T13:05:42.363955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='VIP', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:42.367164Z","iopub.execute_input":"2022-07-28T13:05:42.367515Z","iopub.status.idle":"2022-07-28T13:05:42.585601Z","shell.execute_reply.started":"2022-07-28T13:05:42.367483Z","shell.execute_reply":"2022-07-28T13:05:42.584445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=train_df, x='Age', kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:42.591644Z","iopub.execute_input":"2022-07-28T13:05:42.592007Z","iopub.status.idle":"2022-07-28T13:05:42.980697Z","shell.execute_reply.started":"2022-07-28T13:05:42.591977Z","shell.execute_reply":"2022-07-28T13:05:42.979402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train_df.copy()\ntemp['Age'] = pd.cut(temp['Age'], bins=[0, 12, 20, 40, 120], labels=['Children','Teenage','Adult','Elder'])\nsns.countplot(data=temp, x='Age', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:42.982161Z","iopub.execute_input":"2022-07-28T13:05:42.982636Z","iopub.status.idle":"2022-07-28T13:05:43.231090Z","shell.execute_reply.started":"2022-07-28T13:05:42.982603Z","shell.execute_reply":"2022-07-28T13:05:43.229920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.histplot(data=train_df, x='RoomService', kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:43.232464Z","iopub.execute_input":"2022-07-28T13:05:43.232886Z","iopub.status.idle":"2022-07-28T13:05:43.237859Z","shell.execute_reply.started":"2022-07-28T13:05:43.232853Z","shell.execute_reply":"2022-07-28T13:05:43.236462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp['RoomService'] = pd.cut(temp['RoomService'], bins=[0, 50, 200, 1000, 15000], \n                             labels=['Low RoomService', 'Median RoomService', 'Average RoomService', 'High RoomService'])\nsns.countplot(data=temp, y='RoomService', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:43.239442Z","iopub.execute_input":"2022-07-28T13:05:43.239809Z","iopub.status.idle":"2022-07-28T13:05:43.545482Z","shell.execute_reply.started":"2022-07-28T13:05:43.239778Z","shell.execute_reply":"2022-07-28T13:05:43.544171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.histplot(data=train_df, x='FoodCourt', kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:43.546937Z","iopub.execute_input":"2022-07-28T13:05:43.547273Z","iopub.status.idle":"2022-07-28T13:05:43.552540Z","shell.execute_reply.started":"2022-07-28T13:05:43.547242Z","shell.execute_reply":"2022-07-28T13:05:43.551284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp['FoodCourt'] = pd.cut(temp['FoodCourt'], bins=[0, 100, 400, 2000, 30000],\n                          labels=['Low FoodCourt', 'Median FoodCourt', 'Average FoodCourt', 'High FoodCourt'])\nsns.countplot(data=temp, y='FoodCourt', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:43.554015Z","iopub.execute_input":"2022-07-28T13:05:43.554346Z","iopub.status.idle":"2022-07-28T13:05:43.846951Z","shell.execute_reply.started":"2022-07-28T13:05:43.554317Z","shell.execute_reply":"2022-07-28T13:05:43.845381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.histplot(data=train_df, x='ShoppingMall', kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:43.848624Z","iopub.execute_input":"2022-07-28T13:05:43.849638Z","iopub.status.idle":"2022-07-28T13:05:43.856027Z","shell.execute_reply.started":"2022-07-28T13:05:43.849591Z","shell.execute_reply":"2022-07-28T13:05:43.854245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp['ShoppingMall'] = pd.cut(temp['ShoppingMall'], bins=[0, 100, 400, 2000, 30000],\n                             labels=['Low ShoppingMall', 'Median ShoppingMall', 'Average ShoppingMall', 'High ShoppingMall'])\nsns.countplot(data=temp, y='ShoppingMall', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:43.857774Z","iopub.execute_input":"2022-07-28T13:05:43.858661Z","iopub.status.idle":"2022-07-28T13:05:44.130251Z","shell.execute_reply.started":"2022-07-28T13:05:43.858614Z","shell.execute_reply":"2022-07-28T13:05:44.129105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.histplot(data=train_df, x='Spa')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:44.131873Z","iopub.execute_input":"2022-07-28T13:05:44.133653Z","iopub.status.idle":"2022-07-28T13:05:44.141531Z","shell.execute_reply.started":"2022-07-28T13:05:44.133595Z","shell.execute_reply":"2022-07-28T13:05:44.138735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp['Spa'] = pd.cut(temp['Spa'], bins=[0, 100, 400, 2000, 30000],\n                    labels=['Low Spa', 'Median Spa', 'Average Spa', 'High Spa'])\nsns.countplot(data=temp, x='Spa', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:44.143507Z","iopub.execute_input":"2022-07-28T13:05:44.144407Z","iopub.status.idle":"2022-07-28T13:05:44.399759Z","shell.execute_reply.started":"2022-07-28T13:05:44.144356Z","shell.execute_reply":"2022-07-28T13:05:44.398601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.histplot(data=train_df, x='VRDeck')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:44.401523Z","iopub.execute_input":"2022-07-28T13:05:44.402221Z","iopub.status.idle":"2022-07-28T13:05:44.407670Z","shell.execute_reply.started":"2022-07-28T13:05:44.402176Z","shell.execute_reply":"2022-07-28T13:05:44.406155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp['VRDeck'] = pd.cut(temp['VRDeck'], bins=[0, 100, 400, 2000, 30000],\n                       labels=['Low VRDeck', 'Median VRDeck', 'Average VRDeck', 'High VRDeck'])\nsns.countplot(data=temp, x='VRDeck', hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:44.409243Z","iopub.execute_input":"2022-07-28T13:05:44.410270Z","iopub.status.idle":"2022-07-28T13:05:44.665415Z","shell.execute_reply.started":"2022-07-28T13:05:44.410205Z","shell.execute_reply":"2022-07-28T13:05:44.664294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(train_df.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:44.666884Z","iopub.execute_input":"2022-07-28T13:05:44.667267Z","iopub.status.idle":"2022-07-28T13:05:45.175599Z","shell.execute_reply.started":"2022-07-28T13:05:44.667206Z","shell.execute_reply":"2022-07-28T13:05:45.174186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.corr()['Transported'].sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.177047Z","iopub.execute_input":"2022-07-28T13:05:45.177418Z","iopub.status.idle":"2022-07-28T13:05:45.189813Z","shell.execute_reply.started":"2022-07-28T13:05:45.177386Z","shell.execute_reply":"2022-07-28T13:05:45.188871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['VIP'].mode()[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.191151Z","iopub.execute_input":"2022-07-28T13:05:45.192043Z","iopub.status.idle":"2022-07-28T13:05:45.200859Z","shell.execute_reply.started":"2022-07-28T13:05:45.191992Z","shell.execute_reply":"2022-07-28T13:05:45.199490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA results\n- People from Europa have high chances to be transported\n- People with CryoSleep = True have high chances to be transported\n- CryoSleep is highly correlated with Transported\n- People with 55 Cancri e destination have high chances to be transported\n- Children have high chances to be transported\n- People with high FoodCourt have high chances to be transported\n- People with high ShoppingMall have high chances to transported","metadata":{}},{"cell_type":"markdown","source":"# Data cleaning","metadata":{}},{"cell_type":"code","source":"train_df_utils = {}\nencoded = {}","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.202486Z","iopub.execute_input":"2022-07-28T13:05:45.202982Z","iopub.status.idle":"2022-07-28T13:05:45.211094Z","shell.execute_reply.started":"2022-07-28T13:05:45.202933Z","shell.execute_reply":"2022-07-28T13:05:45.210098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_data(df, target_encoding=True, test=False):\n    global train_df_utils, train_df, encoded\n    df.drop(columns=['PassengerId', 'Name', 'Cabin'], inplace=True, axis=1)\n    if not test:\n        for col in df.columns:\n            if df[col].dtype == 'bool' or df[col].dtype == 'object':\n                train_df_utils[col] = df[col].mode()[0]\n            elif df[col].dtype == 'float64':\n                train_df_utils[col] = df[col].mean()\n    for col in df.columns:\n        if df[col].isna().sum() != 0:\n            df[col].fillna(train_df_utils[col], inplace=True)\n    if not test:\n        for col in df.columns:\n            if df[col].dtype == 'object':\n                encoded[col] = {}\n                unique = train_df[col].unique()\n                unique_values = train_df[col].value_counts()\n                p_cat = {}\n                for cat in unique:\n                    p_cat[cat] = unique_values[cat] / len(train_df)\n                y_mean = train_df['Transported'].mean()\n                y_mean_cat = {}\n                for cat in unique:\n                    y_mean_cat[cat] = train_df[train_df[col] == cat]['Transported'].mean()\n                for cat in unique:\n                    encoded[col][cat] = p_cat[cat] * y_mean_cat[cat] + (1 - p_cat[cat]) * y_mean\n    if target_encoding:\n        for col in df.columns:\n            if df[col].dtype == 'object':\n                df[col] = df[col].apply(lambda x: encoded[col][x])\n    if not target_encoding:\n        df = pd.get_dummies(df, drop_first=True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.212627Z","iopub.execute_input":"2022-07-28T13:05:45.213633Z","iopub.status.idle":"2022-07-28T13:05:45.231084Z","shell.execute_reply.started":"2022-07-28T13:05:45.213583Z","shell.execute_reply":"2022-07-28T13:05:45.229852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = clean_data(train_df)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.232346Z","iopub.execute_input":"2022-07-28T13:05:45.232749Z","iopub.status.idle":"2022-07-28T13:05:45.314559Z","shell.execute_reply.started":"2022-07-28T13:05:45.232716Z","shell.execute_reply":"2022-07-28T13:05:45.313333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/spaceship-titanic/test.csv')\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.321902Z","iopub.execute_input":"2022-07-28T13:05:45.322348Z","iopub.status.idle":"2022-07-28T13:05:45.380481Z","shell.execute_reply.started":"2022-07-28T13:05:45.322311Z","shell.execute_reply":"2022-07-28T13:05:45.379321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['CryoSleep'] = test_df['CryoSleep'].astype(bool)\ntest_df['VIP'] = test_df['VIP'].astype(bool)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.381873Z","iopub.execute_input":"2022-07-28T13:05:45.382336Z","iopub.status.idle":"2022-07-28T13:05:45.391467Z","shell.execute_reply.started":"2022-07-28T13:05:45.382294Z","shell.execute_reply":"2022-07-28T13:05:45.390141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = clean_data(test_df, test=True)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.393143Z","iopub.execute_input":"2022-07-28T13:05:45.393564Z","iopub.status.idle":"2022-07-28T13:05:45.439644Z","shell.execute_reply.started":"2022-07-28T13:05:45.393529Z","shell.execute_reply":"2022-07-28T13:05:45.438615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model defining","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['Transported'])\nY = train_df['Transported']\nX.shape, Y.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.440991Z","iopub.execute_input":"2022-07-28T13:05:45.441378Z","iopub.status.idle":"2022-07-28T13:05:45.454305Z","shell.execute_reply.started":"2022-07-28T13:05:45.441345Z","shell.execute_reply":"2022-07-28T13:05:45.452778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import AdaBoostClassifier, GradientBoostingClassifier, RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, recall_score, precision_score, f1_score, confusion_matrix\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler\nfrom xgboost import XGBClassifier, XGBRFClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.455870Z","iopub.execute_input":"2022-07-28T13:05:45.456394Z","iopub.status.idle":"2022-07-28T13:05:45.923014Z","shell.execute_reply.started":"2022-07-28T13:05:45.456352Z","shell.execute_reply":"2022-07-28T13:05:45.922018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, Y_train, Y_val = train_test_split(X, Y, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.924324Z","iopub.execute_input":"2022-07-28T13:05:45.924746Z","iopub.status.idle":"2022-07-28T13:05:45.933807Z","shell.execute_reply.started":"2022-07-28T13:05:45.924714Z","shell.execute_reply":"2022-07-28T13:05:45.932703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_df = pd.DataFrame(data={'Accuracy': [], 'Precision': [], 'Recall': [], 'F1': []})","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.935574Z","iopub.execute_input":"2022-07-28T13:05:45.936346Z","iopub.status.idle":"2022-07-28T13:05:45.943605Z","shell.execute_reply.started":"2022-07-28T13:05:45.936299Z","shell.execute_reply":"2022-07-28T13:05:45.942499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_performance(model, scores_df, algorithm=\"\"):\n    global X_train, X_val, Y_train, Y_val\n    Y_val_pred = model.predict(X_val)\n    acc = accuracy_score(Y_val, Y_val_pred)\n    prec = precision_score(Y_val, Y_val_pred)\n    rec = recall_score(Y_val, Y_val_pred)\n    f1 = f1_score(Y_val, Y_val_pred)\n    scores_df.loc[algorithm] = [acc, prec, rec, f1]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.946493Z","iopub.execute_input":"2022-07-28T13:05:45.946977Z","iopub.status.idle":"2022-07-28T13:05:45.957309Z","shell.execute_reply.started":"2022-07-28T13:05:45.946922Z","shell.execute_reply":"2022-07-28T13:05:45.956169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"algorithms = {\n    'LogisticRegression': make_pipeline(StandardScaler(), LogisticRegression()),\n    'SVC': make_pipeline(StandardScaler(), SVC()),\n    'AdaBoostClassifier': make_pipeline(StandardScaler(), AdaBoostClassifier(n_estimators=300, random_state=21)),\n    'GradientBoostingClassifier': make_pipeline(StandardScaler(), GradientBoostingClassifier(n_estimators=300, max_depth=5)),\n    'RandomForestClassifier': make_pipeline(StandardScaler(), RandomForestClassifier(n_estimators=300, max_depth=5)),\n    'KNeighborsClassifier': make_pipeline(StandardScaler(), KNeighborsClassifier()),\n    'XGBClassifier': make_pipeline(StandardScaler(), XGBClassifier(n_estimators=300, max_depth=5)),\n    'XGBRFClassifier': make_pipeline(StandardScaler(), XGBRFClassifier(n_estimators=300, max_depth=5)),\n}\n\n# params = {\n#     'LogisticRegression': {}\n#     'SVC': {}\n#     'AdaBoostClassifier': {},\n#     'GradientBoostingClassifier': {},\n#     'RandomForestClassifier': {},\n#     'KNeighborsClassifier': {},\n#     'XGBClassifier': {},\n#     'XGBRFClassifier': {}\n# }","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.958791Z","iopub.execute_input":"2022-07-28T13:05:45.959306Z","iopub.status.idle":"2022-07-28T13:05:45.970512Z","shell.execute_reply.started":"2022-07-28T13:05:45.959274Z","shell.execute_reply":"2022-07-28T13:05:45.969329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for algo in algorithms:\n    algorithms[algo].fit(X_train, Y_train)\n    calculate_performance(algorithms[algo], scores_df, algorithm=algo)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:45.972156Z","iopub.execute_input":"2022-07-28T13:05:45.972515Z","iopub.status.idle":"2022-07-28T13:05:59.400382Z","shell.execute_reply.started":"2022-07-28T13:05:45.972485Z","shell.execute_reply":"2022-07-28T13:05:59.399437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_df.sort_values(by=['Accuracy'], ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:59.401975Z","iopub.execute_input":"2022-07-28T13:05:59.402678Z","iopub.status.idle":"2022-07-28T13:05:59.418097Z","shell.execute_reply.started":"2022-07-28T13:05:59.402637Z","shell.execute_reply":"2022-07-28T13:05:59.416729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_test = algorithms['GradientBoostingClassifier'].predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:59.419822Z","iopub.execute_input":"2022-07-28T13:05:59.420919Z","iopub.status.idle":"2022-07-28T13:05:59.462727Z","shell.execute_reply.started":"2022-07-28T13:05:59.420874Z","shell.execute_reply":"2022-07-28T13:05:59.461759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pass_id = pd.read_csv('../input/spaceship-titanic/test.csv')['PassengerId']\nsolution = pd.DataFrame({'PassengerId': pass_id, 'Transported': Y_test})","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:59.463981Z","iopub.execute_input":"2022-07-28T13:05:59.464538Z","iopub.status.idle":"2022-07-28T13:05:59.491221Z","shell.execute_reply.started":"2022-07-28T13:05:59.464506Z","shell.execute_reply":"2022-07-28T13:05:59.490260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution.to_csv('submission.csv', sep=',', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:05:59.492476Z","iopub.execute_input":"2022-07-28T13:05:59.492981Z","iopub.status.idle":"2022-07-28T13:05:59.510917Z","shell.execute_reply.started":"2022-07-28T13:05:59.492950Z","shell.execute_reply":"2022-07-28T13:05:59.509981Z"},"trusted":true},"execution_count":null,"outputs":[]}]}