{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"****Description****\n\nThe Spaceship Titanic was an interstellar passenger liner launched a month ago. With almost 13,000 passengers on board, the vessel set out on its maiden voyage transporting emigrants from our solar system to three newly habitable exoplanets orbiting nearby stars.\n\nWhile rounding Alpha Centauri en route to its first destination—the torrid 55 Cancri E—the unwary Spaceship Titanic collided with a spacetime anomaly hidden within a dust cloud. Sadly, it met a similar fate as its namesake from 1000 years before. Though the ship stayed intact, almost half of the passengers were transported to an alternate dimension!","metadata":{}},{"cell_type":"markdown","source":"**Import Libraries**","metadata":{}},{"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T08:04:27.187080Z","iopub.execute_input":"2022-07-12T08:04:27.187495Z","iopub.status.idle":"2022-07-12T08:04:27.195580Z","shell.execute_reply.started":"2022-07-12T08:04:27.187461Z","shell.execute_reply":"2022-07-12T08:04:27.194280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.preprocessing import FunctionTransformer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_validate\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import confusion_matrix, plot_confusion_matrix\nfrom sklearn.metrics import roc_curve, roc_auc_score\nfrom sklearn.metrics import accuracy_score, recall_score, precision_score\nimport time\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:27.286720Z","iopub.execute_input":"2022-07-12T08:04:27.287116Z","iopub.status.idle":"2022-07-12T08:04:27.295159Z","shell.execute_reply.started":"2022-07-12T08:04:27.287083Z","shell.execute_reply":"2022-07-12T08:04:27.294041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style=\"ticks\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:27.398511Z","iopub.execute_input":"2022-07-12T08:04:27.399818Z","iopub.status.idle":"2022-07-12T08:04:27.406845Z","shell.execute_reply.started":"2022-07-12T08:04:27.399770Z","shell.execute_reply":"2022-07-12T08:04:27.405366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_path = Path('/kaggle/input/spaceship-titanic/')\ntrain_data = pd.read_csv(input_path / 'train.csv', index_col='PassengerId')\ntest_data = pd.read_csv(input_path / 'test.csv', index_col='PassengerId')\nsubmission = pd.read_csv(input_path / 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:27.526040Z","iopub.execute_input":"2022-07-12T08:04:27.526424Z","iopub.status.idle":"2022-07-12T08:04:27.583124Z","shell.execute_reply.started":"2022-07-12T08:04:27.526392Z","shell.execute_reply":"2022-07-12T08:04:27.581907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:27.638404Z","iopub.execute_input":"2022-07-12T08:04:27.638805Z","iopub.status.idle":"2022-07-12T08:04:27.657059Z","shell.execute_reply.started":"2022-07-12T08:04:27.638775Z","shell.execute_reply":"2022-07-12T08:04:27.655779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:27.666382Z","iopub.execute_input":"2022-07-12T08:04:27.666766Z","iopub.status.idle":"2022-07-12T08:04:27.686221Z","shell.execute_reply.started":"2022-07-12T08:04:27.666737Z","shell.execute_reply":"2022-07-12T08:04:27.684853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Comlumns Description**\n\nPassengerId: A unique Id for each passenger\n\nHomePlanet: The planet the passenger departed from, typically their planet of permanent residence.\n\nCryoSleep: Indicates whether the passenger elected to be put into suspended animation for the duration of the voyage.\n\nCabin :The cabin number where the passenger is staying.\n\nDestination: The planet the passenger will be debarking to.\n\nAge: The age of the passenger.\n\nVIP: Whether the passenger has paid for special VIP service during the voyage.\n\nRoomService: Amount the passenger has billed for room service.\n\nFoodCourt: Amount the passenger has billed at the food court.\n\nShoppingMall: Amount the passenger has billed at the shopping mall.\n\nSpa: Amount the passenger has billed at the spa.\n\nVRDeck: Amount the passenger has billed at the VR deck.\n\nName: The name of the passenger.\n\nTransported: Whether the passenger was transported to another dimension.","metadata":{}},{"cell_type":"markdown","source":"**exploring the data**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,4))\nsns.countplot(data=train_data, x='HomePlanet', hue='Transported')\nplt.title('HomePlanet')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:27.746388Z","iopub.execute_input":"2022-07-12T08:04:27.747661Z","iopub.status.idle":"2022-07-12T08:04:27.973737Z","shell.execute_reply.started":"2022-07-12T08:04:27.747606Z","shell.execute_reply":"2022-07-12T08:04:27.972500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Adding Group column by splitting the index and calculate the group size for each passenger","metadata":{}},{"cell_type":"code","source":"# New feature - Group\ntrain_data['Group'] = [x.split('_')[0] for x in list(train_data.index)]\ntrain_data['Group'] = train_data['Group'].astype(int)\ntest_data['Group'] = [x.split('_')[0] for x in list(test_data.index)]\ntest_data['Group'] = test_data['Group'].astype(int)\n\n# New feature - Group size\ntrain_data['Group_size']=train_data['Group'].map(lambda x: train_data['Group'].value_counts()[x])\ntest_data['Group_size']=test_data['Group'].map(lambda x: test_data['Group'].value_counts()[x])\n\nplt.figure(figsize=(20,4))\nsns.countplot(data=train_data, x='Group_size', hue='Transported')\nplt.title('Group size')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:27.975760Z","iopub.execute_input":"2022-07-12T08:04:27.976229Z","iopub.status.idle":"2022-07-12T08:04:37.523630Z","shell.execute_reply.started":"2022-07-12T08:04:27.976184Z","shell.execute_reply":"2022-07-12T08:04:37.522549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style=\"whitegrid\")\nexpenditure_columns = ['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']\nsns.boxplot(data=train_data[expenditure_columns], orient=\"h\", palette=\"Set2\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:37.525752Z","iopub.execute_input":"2022-07-12T08:04:37.526058Z","iopub.status.idle":"2022-07-12T08:04:37.795304Z","shell.execute_reply.started":"2022-07-12T08:04:37.526031Z","shell.execute_reply":"2022-07-12T08:04:37.794063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Dropping comlumn wich maybe cannot be helpful ","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:37.796748Z","iopub.execute_input":"2022-07-12T08:04:37.797629Z","iopub.status.idle":"2022-07-12T08:04:37.802224Z","shell.execute_reply.started":"2022-07-12T08:04:37.797581Z","shell.execute_reply":"2022-07-12T08:04:37.801147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(columns=['Group','Name'],inplace=True)\ntest_data.drop(columns=['Group','Name'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:37.805425Z","iopub.execute_input":"2022-07-12T08:04:37.806498Z","iopub.status.idle":"2022-07-12T08:04:37.815734Z","shell.execute_reply.started":"2022-07-12T08:04:37.806438Z","shell.execute_reply":"2022-07-12T08:04:37.814644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Columns with nulls values\ntrain_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:37.817516Z","iopub.execute_input":"2022-07-12T08:04:37.817928Z","iopub.status.idle":"2022-07-12T08:04:37.833645Z","shell.execute_reply.started":"2022-07-12T08:04:37.817897Z","shell.execute_reply":"2022-07-12T08:04:37.832550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_data[['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']] = train_data[['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:37.836626Z","iopub.execute_input":"2022-07-12T08:04:37.837377Z","iopub.status.idle":"2022-07-12T08:04:37.845890Z","shell.execute_reply.started":"2022-07-12T08:04:37.837330Z","shell.execute_reply":"2022-07-12T08:04:37.844688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creates new columns from the cabin details :Deck,Num and Side\ntrain_data[['Deck', 'Num', 'Side']] = train_data['Cabin'].str.split('/', expand=True)\ntest_data[['Deck', 'Num', 'Side']] = test_data['Cabin'].str.split('/', expand=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:37.847336Z","iopub.execute_input":"2022-07-12T08:04:37.847852Z","iopub.status.idle":"2022-07-12T08:04:37.878620Z","shell.execute_reply.started":"2022-07-12T08:04:37.847818Z","shell.execute_reply":"2022-07-12T08:04:37.877111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get the total amount the passenger expend\ntrain_data['Expenditure']= train_data[['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']].sum(axis=1)\ntest_data['Expenditure']= test_data[['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:37.880092Z","iopub.execute_input":"2022-07-12T08:04:37.880468Z","iopub.status.idle":"2022-07-12T08:04:37.893029Z","shell.execute_reply.started":"2022-07-12T08:04:37.880435Z","shell.execute_reply":"2022-07-12T08:04:37.891738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(1,15))\nsns.displot(train_data, x=\"Expenditure\", kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:37.894626Z","iopub.execute_input":"2022-07-12T08:04:37.895138Z","iopub.status.idle":"2022-07-12T08:04:38.760249Z","shell.execute_reply.started":"2022-07-12T08:04:37.895105Z","shell.execute_reply":"2022-07-12T08:04:38.758882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['has_expenditure'] = pd.DataFrame(train_data['Expenditure'] > 0).astype(int)\ntest_data['has_expenditure'] = pd.DataFrame(test_data['Expenditure'] > 0).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:38.763787Z","iopub.execute_input":"2022-07-12T08:04:38.764852Z","iopub.status.idle":"2022-07-12T08:04:38.774111Z","shell.execute_reply.started":"2022-07-12T08:04:38.764810Z","shell.execute_reply":"2022-07-12T08:04:38.772866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.groupby('has_expenditure').Transported.mean().plot(kind='barh').set_xlabel('% transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:38.775797Z","iopub.execute_input":"2022-07-12T08:04:38.776283Z","iopub.status.idle":"2022-07-12T08:04:39.000362Z","shell.execute_reply.started":"2022-07-12T08:04:38.776236Z","shell.execute_reply":"2022-07-12T08:04:38.999401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#displaying numerical data over a very wide range of values in a compact way by log scale\nfor col in ['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck','Expenditure']:\n    train_data[col]=np.log(1+train_data[col])\n    test_data[col]=np.log(1+test_data[col])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.001753Z","iopub.execute_input":"2022-07-12T08:04:39.002448Z","iopub.status.idle":"2022-07-12T08:04:39.015443Z","shell.execute_reply.started":"2022-07-12T08:04:39.002411Z","shell.execute_reply":"2022-07-12T08:04:39.014187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(1,15))\nsns.displot(train_data, x=\"Expenditure\", kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.016960Z","iopub.execute_input":"2022-07-12T08:04:39.017693Z","iopub.status.idle":"2022-07-12T08:04:39.452226Z","shell.execute_reply.started":"2022-07-12T08:04:39.017655Z","shell.execute_reply":"2022-07-12T08:04:39.451355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(data=train_data[expenditure_columns], orient=\"h\", palette=\"Set2\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.453636Z","iopub.execute_input":"2022-07-12T08:04:39.454802Z","iopub.status.idle":"2022-07-12T08:04:39.708375Z","shell.execute_reply.started":"2022-07-12T08:04:39.454744Z","shell.execute_reply":"2022-07-12T08:04:39.707118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creates Beans for ages range \ntest_data[\"Age\"].min(), test_data[\"Age\"].max()\nmin_age, max_age = train_data[\"Age\"].min(), train_data[\"Age\"].max()\nbins = np.linspace(min_age,max_age, 7)\nbins\nlabels = [\"0-13 yr olds\", \"13-26 yr olds\", \"26-39 year olds\", \"40-52 yr olds\", \"53-65 yr olds\", \"65-79 yr olds\"]\ntrain_data[\"AgeGroup\"] = pd.cut(train_data[\"Age\"], bins=bins, labels=labels, include_lowest=True)\ntest_data[\"AgeGroup\"] = pd.cut(test_data[\"Age\"], bins=bins, labels=labels, include_lowest=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.709840Z","iopub.execute_input":"2022-07-12T08:04:39.710363Z","iopub.status.idle":"2022-07-12T08:04:39.724700Z","shell.execute_reply.started":"2022-07-12T08:04:39.710326Z","shell.execute_reply":"2022-07-12T08:04:39.723424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.726187Z","iopub.execute_input":"2022-07-12T08:04:39.726519Z","iopub.status.idle":"2022-07-12T08:04:39.746460Z","shell.execute_reply.started":"2022-07-12T08:04:39.726491Z","shell.execute_reply":"2022-07-12T08:04:39.745627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Categorical Data\ncategorical_cols = [col for col in train_data.columns if train_data[col].dtype in [\"object\",\"category\"]]\nnumerical_cols = [col for col in train_data.columns if train_data[col].dtype in [\"float64\",'int64']]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.747679Z","iopub.execute_input":"2022-07-12T08:04:39.748550Z","iopub.status.idle":"2022-07-12T08:04:39.753832Z","shell.execute_reply.started":"2022-07-12T08:04:39.748498Z","shell.execute_reply":"2022-07-12T08:04:39.752919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pipelines are a simple way to keep your data preprocessing and modeling code organized.\n#Specifically, a pipeline bundles preprocessing and modeling steps so you can use the whole\n#bundle as if it were a single step. \n# Preprocessing for numerical data\nnumerical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n]) \n\ndef to_int(x):\n    return pd.DataFrame(x).astype(int)\n\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(sparse=False,handle_unknown='ignore'))\n])\n\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols),\n        \n    ])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.755160Z","iopub.execute_input":"2022-07-12T08:04:39.755669Z","iopub.status.idle":"2022-07-12T08:04:39.768209Z","shell.execute_reply.started":"2022-07-12T08:04:39.755637Z","shell.execute_reply":"2022-07-12T08:04:39.766890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#In scikit-learn a random split into training and test sets can be quickly \n#computed with the train_test_split helper function.\nX = train_data.copy()\ny = X.pop('Transported')\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.769912Z","iopub.execute_input":"2022-07-12T08:04:39.771295Z","iopub.status.idle":"2022-07-12T08:04:39.794242Z","shell.execute_reply.started":"2022-07-12T08:04:39.771246Z","shell.execute_reply":"2022-07-12T08:04:39.793012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nXGBoost, which stands for Extreme Gradient Boosting, is a scalable, distributed gradient-boosted \ndecision tree (GBDT) machine learning library. It provides parallel tree boosting and is\nthe leading machine learning library for regression, classification, and ranking problems.\n\"\"\"\nmodel = XGBClassifier(n_estimators=100, colsample_bytree = 0.7, max_depth= 5, learning_rate= 0.01\n                      #,tree_method='gpu_hist', predictor=\"gpu_predictor\"\n                     )\nxgb_pipeline = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', model)\n                             ])\n\n# # Preprocessing of training data, fit model \nxgb_pipeline.fit(X_train, y_train)\n\n# # Preprocessing of validation data, get predictions\npreds = xgb_pipeline.predict(X_valid)\n\n# # Evaluate the model\nscore = accuracy_score(y_valid, preds)\nprint('accuracy_score:', score)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:04:39.795761Z","iopub.execute_input":"2022-07-12T08:04:39.796086Z","iopub.status.idle":"2022-07-12T08:05:27.901264Z","shell.execute_reply.started":"2022-07-12T08:04:39.796046Z","shell.execute_reply":"2022-07-12T08:05:27.900053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_test = xgb_pipeline.predict(test_data) \nsubmission['Transported'] = preds_test.astype(bool)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:05:27.902612Z","iopub.execute_input":"2022-07-12T08:05:27.902947Z","iopub.status.idle":"2022-07-12T08:05:28.265978Z","shell.execute_reply.started":"2022-07-12T08:05:27.902917Z","shell.execute_reply":"2022-07-12T08:05:28.264673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:05:28.267917Z","iopub.execute_input":"2022-07-12T08:05:28.268750Z","iopub.status.idle":"2022-07-12T08:05:28.280882Z","shell.execute_reply.started":"2022-07-12T08:05:28.268706Z","shell.execute_reply":"2022-07-12T08:05:28.279584Z"},"trusted":true},"execution_count":null,"outputs":[]}]}