{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T11:46:16.926388Z","iopub.execute_input":"2022-08-03T11:46:16.927385Z","iopub.status.idle":"2022-08-03T11:46:16.963614Z","shell.execute_reply.started":"2022-08-03T11:46:16.927228Z","shell.execute_reply":"2022-08-03T11:46:16.962083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv')\ntrain_data.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:16.966386Z","iopub.execute_input":"2022-08-03T11:46:16.966919Z","iopub.status.idle":"2022-08-03T11:46:17.061179Z","shell.execute_reply.started":"2022-08-03T11:46:16.966866Z","shell.execute_reply":"2022-08-03T11:46:17.059916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.063027Z","iopub.execute_input":"2022-08-03T11:46:17.063584Z","iopub.status.idle":"2022-08-03T11:46:17.071804Z","shell.execute_reply.started":"2022-08-03T11:46:17.063531Z","shell.execute_reply":"2022-08-03T11:46:17.070852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.074589Z","iopub.execute_input":"2022-08-03T11:46:17.075472Z","iopub.status.idle":"2022-08-03T11:46:17.093277Z","shell.execute_reply.started":"2022-08-03T11:46:17.075429Z","shell.execute_reply":"2022-08-03T11:46:17.092418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.094631Z","iopub.execute_input":"2022-08-03T11:46:17.095137Z","iopub.status.idle":"2022-08-03T11:46:17.135608Z","shell.execute_reply.started":"2022-08-03T11:46:17.095104Z","shell.execute_reply":"2022-08-03T11:46:17.134613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.136961Z","iopub.execute_input":"2022-08-03T11:46:17.137325Z","iopub.status.idle":"2022-08-03T11:46:17.179870Z","shell.execute_reply.started":"2022-08-03T11:46:17.137291Z","shell.execute_reply":"2022-08-03T11:46:17.178860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_data.drop(['Transported'], axis=1)\ny = train_data.Transported","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.181381Z","iopub.execute_input":"2022-08-03T11:46:17.181704Z","iopub.status.idle":"2022-08-03T11:46:17.191125Z","shell.execute_reply.started":"2022-08-03T11:46:17.181675Z","shell.execute_reply":"2022-08-03T11:46:17.189899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X['GroupId'] = X['PassengerId'].map(lambda x: int(x.split('_')[0]))\nX['NumberInGroup'] = X['PassengerId'].map(lambda x: int(x.split('_')[1]))\nX = X.drop(['PassengerId'], axis=1)\nX.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.192857Z","iopub.execute_input":"2022-08-03T11:46:17.193369Z","iopub.status.idle":"2022-08-03T11:46:17.242580Z","shell.execute_reply.started":"2022-08-03T11:46:17.193295Z","shell.execute_reply":"2022-08-03T11:46:17.241312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X['CabinDeck'] = X['Cabin'].map(lambda x: x.split('/')[0] if not pd.isna(x) else np.nan)\nX['CabinNum'] = X['Cabin'].map(lambda x: int(x.split('/')[1]) if not pd.isna(x) else np.nan)\nX['CabinSide'] = X['Cabin'].map(lambda x: x.split('/')[2] if not pd.isna(x) else np.nan)\nX = X.drop(['Cabin'], axis=1)\nX.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.243581Z","iopub.execute_input":"2022-08-03T11:46:17.243896Z","iopub.status.idle":"2022-08-03T11:46:17.308605Z","shell.execute_reply.started":"2022-08-03T11:46:17.243867Z","shell.execute_reply":"2022-08-03T11:46:17.307427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_lengths = X.Name.map(lambda x: len(list(x.split())) if not pd.isna(x) else -1)\nname_lengths.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.313886Z","iopub.execute_input":"2022-08-03T11:46:17.314237Z","iopub.status.idle":"2022-08-03T11:46:17.337023Z","shell.execute_reply.started":"2022-08-03T11:46:17.314207Z","shell.execute_reply":"2022-08-03T11:46:17.335838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X['FirstName'] = X.Name.map(lambda x: x.split()[0] if not pd.isna(x) else np.nan)\nX['LastName'] = X.Name.map(lambda x: x.split()[1] if not pd.isna(x) else np.nan)\nX = X.drop(['Name'], axis=1)\nX.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.338702Z","iopub.execute_input":"2022-08-03T11:46:17.339493Z","iopub.status.idle":"2022-08-03T11:46:17.390655Z","shell.execute_reply.started":"2022-08-03T11:46:17.339437Z","shell.execute_reply":"2022-08-03T11:46:17.389760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X.HomePlanet.value_counts()\n# X.CryoSleep.value_counts()\n# X.CabinDeck.value_counts()\n# X.CabinNum.value_counts()\n# X.CabinSide.value_counts()\n# X.Destination.value_counts()\n# X.VIP.value_counts()\nX.FirstName.value_counts()\n# X.LastName.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.391664Z","iopub.execute_input":"2022-08-03T11:46:17.391976Z","iopub.status.idle":"2022-08-03T11:46:17.406303Z","shell.execute_reply.started":"2022-08-03T11:46:17.391948Z","shell.execute_reply":"2022-08-03T11:46:17.404993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\nX = X.fillna({'HomePlanet' : 'Earth', 'CryoSleep': False, 'CabinDeck': 'F', 'CabinNum': 82, 'CabinSide': 'S', \\\n              'Destination': 'TRAPPIST-1e', 'Age': X.Age.median(), 'VIP': False, 'RoomService': X.RoomService.median(), \\\n              'FoodCourt': X.FoodCourt.median(), 'ShoppingMall': X.ShoppingMall.median(), 'Spa': X.Spa.median(), 'VRDeck': X.VRDeck.median()})\nX.FirstName.fillna(random.choice(X.FirstName[X.FirstName.notna()]), inplace=True)\nX.LastName.fillna(random.choice(X.LastName[X.LastName.notna()]), inplace=True)\n\nX.CryoSleep = X.CryoSleep.astype(int)\nX.VIP = X.VIP.astype(int)\ny = y.astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.407956Z","iopub.execute_input":"2022-08-03T11:46:17.409048Z","iopub.status.idle":"2022-08-03T11:46:17.445933Z","shell.execute_reply.started":"2022-08-03T11:46:17.408991Z","shell.execute_reply":"2022-08-03T11:46:17.444994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.447099Z","iopub.execute_input":"2022-08-03T11:46:17.447955Z","iopub.status.idle":"2022-08-03T11:46:17.463228Z","shell.execute_reply.started":"2022-08-03T11:46:17.447918Z","shell.execute_reply":"2022-08-03T11:46:17.462248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:17.464710Z","iopub.execute_input":"2022-08-03T11:46:17.465512Z","iopub.status.idle":"2022-08-03T11:46:18.816649Z","shell.execute_reply.started":"2022-08-03T11:46:17.465477Z","shell.execute_reply":"2022-08-03T11:46:18.815291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier\n\ncat_features_index = np.where(X.dtypes != float)[0]\n\nmodel = CatBoostClassifier(eval_metric='Accuracy', use_best_model=True, random_seed=42, iterations=100)\nmodel.fit(X_train, y_train, cat_features=cat_features_index, eval_set=(X_val,y_val), plot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:18.818762Z","iopub.execute_input":"2022-08-03T11:46:18.819445Z","iopub.status.idle":"2022-08-03T11:46:20.556703Z","shell.execute_reply.started":"2022-08-03T11:46:18.819392Z","shell.execute_reply":"2022-08-03T11:46:20.555551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:20.558376Z","iopub.execute_input":"2022-08-03T11:46:20.558749Z","iopub.status.idle":"2022-08-03T11:46:20.592679Z","shell.execute_reply.started":"2022-08-03T11:46:20.558708Z","shell.execute_reply":"2022-08-03T11:46:20.591488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:20.594553Z","iopub.execute_input":"2022-08-03T11:46:20.595268Z","iopub.status.idle":"2022-08-03T11:46:20.610048Z","shell.execute_reply.started":"2022-08-03T11:46:20.595218Z","shell.execute_reply":"2022-08-03T11:46:20.608904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_lengths = test_data.Name.map(lambda x: len(list(x.split())) if not pd.isna(x) else -1)\nname_lengths.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:20.612490Z","iopub.execute_input":"2022-08-03T11:46:20.612931Z","iopub.status.idle":"2022-08-03T11:46:20.637177Z","shell.execute_reply.started":"2022-08-03T11:46:20.612879Z","shell.execute_reply":"2022-08-03T11:46:20.635827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['GroupId'] = test_data['PassengerId'].map(lambda x: int(x.split('_')[0]))\ntest_data['NumberInGroup'] = test_data['PassengerId'].map(lambda x: int(x.split('_')[1]))\npassenger_ids = test_data.PassengerId\ntest_data = test_data.drop(['PassengerId'], axis=1)\n\ntest_data['CabinDeck'] = test_data['Cabin'].map(lambda x: x.split('/')[0] if not pd.isna(x) else np.nan)\ntest_data['CabinNum'] = test_data['Cabin'].map(lambda x: int(x.split('/')[1]) if not pd.isna(x) else np.nan)\ntest_data['CabinSide'] = test_data['Cabin'].map(lambda x: x.split('/')[2] if not pd.isna(x) else np.nan)\ntest_data = test_data.drop(['Cabin'], axis=1)\n\ntest_data['FirstName'] = test_data.Name.map(lambda x: x.split()[0] if not pd.isna(x) else np.nan)\ntest_data['LastName'] = test_data.Name.map(lambda x: x.split()[1] if not pd.isna(x) else np.nan)\ntest_data = test_data.drop(['Name'], axis=1)\n\ntest_data = test_data.fillna({'HomePlanet' : 'Earth', 'CryoSleep': False, 'CabinDeck': 'F', 'CabinNum': 82, 'CabinSide': 'S', \\\n              'Destination': 'TRAPPIST-1e', 'Age': X.Age.median(), 'VIP': False, 'RoomService': X.RoomService.median(), \\\n              'FoodCourt': X.FoodCourt.median(), 'ShoppingMall': X.ShoppingMall.median(), 'Spa': X.Spa.median(), 'VRDeck': X.VRDeck.median()})\n\ntest_data.FirstName.fillna(random.choice(X.FirstName[X.FirstName.notna()]), inplace=True)\ntest_data.LastName.fillna(random.choice(X.LastName[X.LastName.notna()]), inplace=True)\n\ntest_data.CryoSleep = test_data.CryoSleep.astype(int)\ntest_data.VIP = test_data.VIP.astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:20.639670Z","iopub.execute_input":"2022-08-03T11:46:20.640164Z","iopub.status.idle":"2022-08-03T11:46:20.731807Z","shell.execute_reply.started":"2022-08-03T11:46:20.640118Z","shell.execute_reply":"2022-08-03T11:46:20.730830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:20.733395Z","iopub.execute_input":"2022-08-03T11:46:20.734018Z","iopub.status.idle":"2022-08-03T11:46:20.745011Z","shell.execute_reply.started":"2022-08-03T11:46:20.733984Z","shell.execute_reply":"2022-08-03T11:46:20.743571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_data)\npred = pred.astype(bool)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:20.746902Z","iopub.execute_input":"2022-08-03T11:46:20.747276Z","iopub.status.idle":"2022-08-03T11:46:20.774228Z","shell.execute_reply.started":"2022-08-03T11:46:20.747242Z","shell.execute_reply":"2022-08-03T11:46:20.772884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'Transported': pred}, index=passenger_ids)\nsubmission.to_csv('catboost.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:20.775870Z","iopub.execute_input":"2022-08-03T11:46:20.776248Z","iopub.status.idle":"2022-08-03T11:46:20.790878Z","shell.execute_reply.started":"2022-08-03T11:46:20.776212Z","shell.execute_reply":"2022-08-03T11:46:20.789809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = model.feature_importances_\nfeature_importances_df = pd.DataFrame({'features': list(X_train), \n                                       'feature_importances': feature_importances})\nfeature_importances_df.sort_values('feature_importances', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:20.792557Z","iopub.execute_input":"2022-08-03T11:46:20.792912Z","iopub.status.idle":"2022-08-03T11:46:20.808905Z","shell.execute_reply.started":"2022-08-03T11:46:20.792879Z","shell.execute_reply":"2022-08-03T11:46:20.807744Z"},"trusted":true},"execution_count":null,"outputs":[]}]}