{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The Notebook will contain the following sections :\n\n - Exploratory Data analysis\n \n - Data Visualiztion\n \n - Data Preprocessing\n \n - ML Section","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.758018Z","iopub.execute_input":"2022-07-09T09:33:27.758417Z","iopub.status.idle":"2022-07-09T09:33:27.765567Z","shell.execute_reply.started":"2022-07-09T09:33:27.758387Z","shell.execute_reply":"2022-07-09T09:33:27.763354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom termcolor import colored\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.769270Z","iopub.execute_input":"2022-07-09T09:33:27.770200Z","iopub.status.idle":"2022-07-09T09:33:27.791574Z","shell.execute_reply.started":"2022-07-09T09:33:27.770168Z","shell.execute_reply":"2022-07-09T09:33:27.790755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir('../input/spaceship-titanic')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.793081Z","iopub.execute_input":"2022-07-09T09:33:27.793767Z","iopub.status.idle":"2022-07-09T09:33:27.881314Z","shell.execute_reply.started":"2022-07-09T09:33:27.793714Z","shell.execute_reply":"2022-07-09T09:33:27.879426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('./train.csv')\ntest = pd.read_csv('./test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.882489Z","iopub.status.idle":"2022-07-09T09:33:27.882878Z","shell.execute_reply.started":"2022-07-09T09:33:27.882731Z","shell.execute_reply":"2022-07-09T09:33:27.882745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.883838Z","iopub.status.idle":"2022-07-09T09:33:27.884233Z","shell.execute_reply.started":"2022-07-09T09:33:27.884086Z","shell.execute_reply":"2022-07-09T09:33:27.884100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.884952Z","iopub.status.idle":"2022-07-09T09:33:27.885326Z","shell.execute_reply.started":"2022-07-09T09:33:27.885191Z","shell.execute_reply":"2022-07-09T09:33:27.885205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Number of rows and columns in train and test files.","metadata":{}},{"cell_type":"code","source":"print(colored('Train data','magenta',attrs=['bold','underline']))\nprint(colored(f'Number of rows in training data is : {df.shape[0]}','blue'))\nprint(colored(f'Number of columns in training data is : {df.shape[1]}\\n','blue'))\nprint(colored('Test data','magenta',attrs=['bold','underline']))\nprint(colored(f'Number of rows in test data is : {test.shape[0]}','blue'))\nprint(colored(f'Number of columns in test data is : {test.shape[1]}\\n','blue'))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.886288Z","iopub.status.idle":"2022-07-09T09:33:27.886653Z","shell.execute_reply.started":"2022-07-09T09:33:27.886478Z","shell.execute_reply":"2022-07-09T09:33:27.886495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.888107Z","iopub.status.idle":"2022-07-09T09:33:27.888885Z","shell.execute_reply.started":"2022-07-09T09:33:27.888700Z","shell.execute_reply":"2022-07-09T09:33:27.888719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking for missing values in the train and test datasets.","metadata":{}},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.890157Z","iopub.status.idle":"2022-07-09T09:33:27.890539Z","shell.execute_reply.started":"2022-07-09T09:33:27.890335Z","shell.execute_reply":"2022-07-09T09:33:27.890352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.891937Z","iopub.status.idle":"2022-07-09T09:33:27.892232Z","shell.execute_reply.started":"2022-07-09T09:33:27.892083Z","shell.execute_reply":"2022-07-09T09:33:27.892099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Conclusion both the train and test dataset contain missing values and need to be handled properly.","metadata":{}},{"cell_type":"markdown","source":"# Data Visualization for individual features.(Univariate Analysis)","metadata":{}},{"cell_type":"markdown","source":"**HomePlanet**","metadata":{}},{"cell_type":"code","source":"sns.set()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.946649Z","iopub.execute_input":"2022-07-09T09:33:27.946966Z","iopub.status.idle":"2022-07-09T09:33:27.951485Z","shell.execute_reply.started":"2022-07-09T09:33:27.946942Z","shell.execute_reply":"2022-07-09T09:33:27.950740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For homeplanet in training\nprint(colored('Home Planet Count Train data','red', attrs=['bold','underline']))\nprint(colored(df['HomePlanet'].value_counts(),'blue',attrs=['bold']))\nfig,ax = plt.subplots(figsize=(10,10))\nsns.countplot(df['HomePlanet'],data=df,order= df['HomePlanet'].value_counts(ascending=False).index)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:27.953117Z","iopub.execute_input":"2022-07-09T09:33:27.953896Z","iopub.status.idle":"2022-07-09T09:33:28.103267Z","shell.execute_reply.started":"2022-07-09T09:33:27.953867Z","shell.execute_reply":"2022-07-09T09:33:28.102171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For homeplanet in test\nfig,ax = plt.subplots(figsize=(10,10))\nprint(colored('Home Planet Count Test data','red', attrs=['bold','underline']))\nprint(colored(test['HomePlanet'].value_counts(),'blue',attrs=['bold']))\nsns.countplot(test['HomePlanet'],data=df,order= test['HomePlanet'].value_counts(ascending=False).index)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:28.105294Z","iopub.execute_input":"2022-07-09T09:33:28.106364Z","iopub.status.idle":"2022-07-09T09:33:28.253240Z","shell.execute_reply.started":"2022-07-09T09:33:28.106334Z","shell.execute_reply":"2022-07-09T09:33:28.252496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bivariate Analysis.","metadata":{}},{"cell_type":"code","source":"# Home planet vs Transported column train data.\nfig,ax = plt.subplots(figsize=(10,10))\nprint(colored(df[['HomePlanet','Transported']].value_counts(),'green'))\nsns.countplot('HomePlanet',hue='Transported',data=df,order=df['HomePlanet'].value_counts(ascending=False).index, ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:28.254311Z","iopub.execute_input":"2022-07-09T09:33:28.255351Z","iopub.status.idle":"2022-07-09T09:33:28.443649Z","shell.execute_reply.started":"2022-07-09T09:33:28.255321Z","shell.execute_reply":"2022-07-09T09:33:28.441962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Europa and Mars seems to have a higher chance of a passenger being transported.\n\nEarth seems to have a lower chances of being transported .","metadata":{}},{"cell_type":"markdown","source":"**CryoSleep**","metadata":{}},{"cell_type":"code","source":"# Cryosleep count\nprint(colored('People in Cryo Sleep Count Train data','red', attrs=['bold','underline']))\nprint(colored(df['CryoSleep'].value_counts(),'cyan',attrs=['bold']))\nfig, ax = plt.subplots(figsize=(10,10))\nsns.countplot('CryoSleep',data=df).set(title='Cryo Sleep count on train data')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:28.445737Z","iopub.execute_input":"2022-07-09T09:33:28.446702Z","iopub.status.idle":"2022-07-09T09:33:28.628006Z","shell.execute_reply.started":"2022-07-09T09:33:28.446671Z","shell.execute_reply":"2022-07-09T09:33:28.626970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(colored('People in Cryo Sleep Count Test data','red', attrs=['bold','underline']))\nprint(colored(test['CryoSleep'].value_counts(),'cyan',attrs=['bold']))\nfig, ax = plt.subplots(figsize=(10,10))\nsns.countplot('CryoSleep',data=test).set(title='Cryo Sleep count on test data')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:28.629245Z","iopub.execute_input":"2022-07-09T09:33:28.629521Z","iopub.status.idle":"2022-07-09T09:33:28.782157Z","shell.execute_reply.started":"2022-07-09T09:33:28.629493Z","shell.execute_reply":"2022-07-09T09:33:28.780843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bivariate Analysis.[CryoSleep vs Transported]","metadata":{}},{"cell_type":"code","source":"print(colored(df[['CryoSleep','Transported']].value_counts(),'green'))\nsns.countplot('CryoSleep',hue='Transported',data=df)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:28.783615Z","iopub.execute_input":"2022-07-09T09:33:28.783976Z","iopub.status.idle":"2022-07-09T09:33:28.985604Z","shell.execute_reply.started":"2022-07-09T09:33:28.783939Z","shell.execute_reply":"2022-07-09T09:33:28.984462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on visualization we can see more people in cryosleep were transported so we can say they ar positively correlated.","metadata":{}},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:28.987038Z","iopub.execute_input":"2022-07-09T09:33:28.987314Z","iopub.status.idle":"2022-07-09T09:33:28.994661Z","shell.execute_reply.started":"2022-07-09T09:33:28.987286Z","shell.execute_reply":"2022-07-09T09:33:28.993681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Destination**","metadata":{}},{"cell_type":"code","source":"print(colored(\"People's Chosen Destination Count Train data\",'red', attrs=['bold','underline']))\nprint(colored(df['Destination'].value_counts(),'cyan',attrs=['bold']))\nfig, ax = plt.subplots(figsize=(10,10))\nsns.countplot('Destination',data=df,order=df['Destination'].value_counts(ascending=False).index).set(title='Destiantion count on train data')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:28.996238Z","iopub.execute_input":"2022-07-09T09:33:28.996731Z","iopub.status.idle":"2022-07-09T09:33:29.152195Z","shell.execute_reply.started":"2022-07-09T09:33:28.996701Z","shell.execute_reply":"2022-07-09T09:33:29.151330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(colored(\"People's Chosen Destination Count Test data\",'red', attrs=['bold','underline']))\nprint(colored(test['Destination'].value_counts(),'cyan',attrs=['bold']))\nfig, ax = plt.subplots(figsize=(10,10))\nsns.countplot('Destination',data=test,order=test['Destination'].value_counts(ascending=False).index).set(title='Destiantion count on test data')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:29.153500Z","iopub.execute_input":"2022-07-09T09:33:29.153931Z","iopub.status.idle":"2022-07-09T09:33:29.336765Z","shell.execute_reply.started":"2022-07-09T09:33:29.153900Z","shell.execute_reply":"2022-07-09T09:33:29.336040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bivariate Analysis training. [Destination vs Transported]","metadata":{}},{"cell_type":"code","source":"print(colored(df[['Destination','Transported']].value_counts(),'green',attrs=['bold']))\nsns.countplot('Destination',hue='Transported',data=df)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:29.340533Z","iopub.execute_input":"2022-07-09T09:33:29.342129Z","iopub.status.idle":"2022-07-09T09:33:29.523788Z","shell.execute_reply.started":"2022-07-09T09:33:29.342099Z","shell.execute_reply":"2022-07-09T09:33:29.522389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on the graph we can see that Trappist-1e shows negative correlation with transported where as the other two planets Pso & 55 Cancri shows positive correlation with transported where cancri ahs the strongest relation.","metadata":{}},{"cell_type":"markdown","source":"**AGE**","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15,10))\nsns.histplot(x='Age',hue='Transported',data=df,kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:29.525690Z","iopub.execute_input":"2022-07-09T09:33:29.525954Z","iopub.status.idle":"2022-07-09T09:33:30.041224Z","shell.execute_reply.started":"2022-07-09T09:33:29.525928Z","shell.execute_reply":"2022-07-09T09:33:30.040275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bivariate Analysis.[Age vs Transported]","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(10,10))\nsns.boxplot('Transported','Age',hue='Transported',data=df,ax=ax)\ndf['Age'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:30.042464Z","iopub.execute_input":"2022-07-09T09:33:30.042720Z","iopub.status.idle":"2022-07-09T09:33:30.293348Z","shell.execute_reply.started":"2022-07-09T09:33:30.042690Z","shell.execute_reply":"2022-07-09T09:33:30.292330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Age distribution of prople who were succesfully transported\nTransport_1 = df[df['Transported']==True]['Age']\nprint(colored('Age distribution of people who were succesfully transported','grey',attrs=['bold']))\nprint(colored(Transport_1.describe(),'magenta',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:30.295631Z","iopub.execute_input":"2022-07-09T09:33:30.296061Z","iopub.status.idle":"2022-07-09T09:33:30.312737Z","shell.execute_reply.started":"2022-07-09T09:33:30.296024Z","shell.execute_reply":"2022-07-09T09:33:30.311417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Transport_0 = df[df['Transported']==False]['Age']\nprint(colored('Age distribution of people who were not succesfully transported','grey',attrs=['bold']))\nprint(colored(Transport_0.describe(),'magenta',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:30.314078Z","iopub.execute_input":"2022-07-09T09:33:30.314395Z","iopub.status.idle":"2022-07-09T09:33:30.324692Z","shell.execute_reply.started":"2022-07-09T09:33:30.314364Z","shell.execute_reply":"2022-07-09T09:33:30.324012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on the above info we can see that younger people have a higher chance of survival.","metadata":{}},{"cell_type":"markdown","source":"**VIP**","metadata":{}},{"cell_type":"code","source":"print(colored('VIP Statstics','blue',attrs=['bold','underline']))\nprint(colored(df[['VIP','Transported']].value_counts(),'grey',attrs=['bold']))\nfig,ax = plt.subplots(figsize=(15,10))\nsns.countplot('VIP',hue='Transported',data=df,ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:30.326202Z","iopub.execute_input":"2022-07-09T09:33:30.326594Z","iopub.status.idle":"2022-07-09T09:33:30.551339Z","shell.execute_reply.started":"2022-07-09T09:33:30.326563Z","shell.execute_reply":"2022-07-09T09:33:30.550358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:30.552535Z","iopub.execute_input":"2022-07-09T09:33:30.552811Z","iopub.status.idle":"2022-07-09T09:33:30.575826Z","shell.execute_reply.started":"2022-07-09T09:33:30.552782Z","shell.execute_reply":"2022-07-09T09:33:30.574420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**RoomService FoodCourt ShoppingMall Spa VRDeck**","metadata":{}},{"cell_type":"code","source":"# Boxplot and histogram for each features\nfig, ax = plt.subplots(1,2,figsize=(20,9))\nfig.suptitle('RoomService histogram and boxplots')\n# ax[0] denotes 1st row 1st column and ax[2] denotes 1nd row 2nd column from subplots \nsns.histplot(x='RoomService',hue='Transported',data=df,kde=True,ax=ax[0],element='step')\nsns.boxplot('Transported','RoomService',data=df,ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:30.577395Z","iopub.execute_input":"2022-07-09T09:33:30.577768Z","iopub.status.idle":"2022-07-09T09:33:31.128302Z","shell.execute_reply.started":"2022-07-09T09:33:30.577740Z","shell.execute_reply":"2022-07-09T09:33:31.127150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,9))\nfig.suptitle('FoodCourt histogram and boxplots')\nsns.histplot(x='FoodCourt',hue='Transported',data=df,kde=True,ax=ax[0],element='step')\nsns.boxplot('Transported','FoodCourt',data=df,ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:31.129334Z","iopub.execute_input":"2022-07-09T09:33:31.129568Z","iopub.status.idle":"2022-07-09T09:33:31.537186Z","shell.execute_reply.started":"2022-07-09T09:33:31.129546Z","shell.execute_reply":"2022-07-09T09:33:31.536262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,9))\nfig.suptitle('ShoppingMall histogram and boxplots')\nsns.histplot(x='ShoppingMall',hue='Transported',data=df,kde=True,ax=ax[0],element='step')\nsns.boxplot('Transported','ShoppingMall',data=df,ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:31.538891Z","iopub.execute_input":"2022-07-09T09:33:31.539271Z","iopub.status.idle":"2022-07-09T09:33:32.296478Z","shell.execute_reply.started":"2022-07-09T09:33:31.539230Z","shell.execute_reply":"2022-07-09T09:33:32.295331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,9))\nfig.suptitle('Spa histogram and boxplots')\nsns.histplot(x='Spa',hue='Transported',data=df,kde=True,ax=ax[0],element='step')\nsns.boxplot('Transported','Spa',data=df,ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:32.297645Z","iopub.execute_input":"2022-07-09T09:33:32.297887Z","iopub.status.idle":"2022-07-09T09:33:32.760864Z","shell.execute_reply.started":"2022-07-09T09:33:32.297864Z","shell.execute_reply":"2022-07-09T09:33:32.759802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,9))\nfig.suptitle('VRDeck histogram and boxplots')\nsns.histplot(x='VRDeck',hue='Transported',data=df,kde=True,ax=ax[0],element='step')\nsns.boxplot('Transported','VRDeck',data=df,ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:32.762337Z","iopub.execute_input":"2022-07-09T09:33:32.762617Z","iopub.status.idle":"2022-07-09T09:33:33.285845Z","shell.execute_reply.started":"2022-07-09T09:33:32.762591Z","shell.execute_reply":"2022-07-09T09:33:33.284764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Age vs Money spend on amenities**","metadata":{}},{"cell_type":"code","source":"fig,ax=plt.subplots(figsize=(10,5))\nplt.suptitle(\"Age vs Money spend on RoomService\")\nsns.scatterplot('Age','RoomService',hue='Transported',data=df,ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:33.287265Z","iopub.execute_input":"2022-07-09T09:33:33.287556Z","iopub.status.idle":"2022-07-09T09:33:33.875947Z","shell.execute_reply.started":"2022-07-09T09:33:33.287529Z","shell.execute_reply":"2022-07-09T09:33:33.874759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax=plt.subplots(figsize=(10,5))\nplt.suptitle(\"Age vs Money spend on FoodCourt\")\nsns.scatterplot('Age','FoodCourt',hue='Transported',data=df,ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:33.876931Z","iopub.execute_input":"2022-07-09T09:33:33.877199Z","iopub.status.idle":"2022-07-09T09:33:34.307753Z","shell.execute_reply.started":"2022-07-09T09:33:33.877172Z","shell.execute_reply":"2022-07-09T09:33:34.306685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax=plt.subplots(figsize=(10,5))\nplt.suptitle(\"Age vs Money spend on ShoppingMall\")\nsns.scatterplot('Age','ShoppingMall',hue='Transported',data=df,ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:34.309071Z","iopub.execute_input":"2022-07-09T09:33:34.309784Z","iopub.status.idle":"2022-07-09T09:33:34.722278Z","shell.execute_reply.started":"2022-07-09T09:33:34.309757Z","shell.execute_reply":"2022-07-09T09:33:34.721130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax=plt.subplots(figsize=(10,5))\nplt.suptitle(\"Age vs Money spend on Spa\")\nsns.scatterplot('Age','Spa',hue='Transported',data=df,ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:34.723627Z","iopub.execute_input":"2022-07-09T09:33:34.723907Z","iopub.status.idle":"2022-07-09T09:33:35.151147Z","shell.execute_reply.started":"2022-07-09T09:33:34.723880Z","shell.execute_reply":"2022-07-09T09:33:35.150013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax=plt.subplots(figsize=(10,5))\nplt.suptitle(\"Age vs Money spend on VRDeck\")\nsns.scatterplot('Age','VRDeck',hue='Transported',data=df,ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:35.152555Z","iopub.execute_input":"2022-07-09T09:33:35.152886Z","iopub.status.idle":"2022-07-09T09:33:35.574466Z","shell.execute_reply.started":"2022-07-09T09:33:35.152857Z","shell.execute_reply":"2022-07-09T09:33:35.573461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Creating a new column to add up all the values of the amenities such as RoomService FoodCourt ShoppingMall Spa VRDeck**","metadata":{}},{"cell_type":"code","source":"Amenities = ['RoomService','FoodCourt','ShoppingMall','Spa','VRDeck'] \n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:35.575599Z","iopub.execute_input":"2022-07-09T09:33:35.575825Z","iopub.status.idle":"2022-07-09T09:33:35.580404Z","shell.execute_reply.started":"2022-07-09T09:33:35.575803Z","shell.execute_reply":"2022-07-09T09:33:35.579471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['TotalSpend'] = df[Amenities].sum(axis=1)\ntest['TotalSpend'] = test[Amenities].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:35.588466Z","iopub.execute_input":"2022-07-09T09:33:35.589169Z","iopub.status.idle":"2022-07-09T09:33:35.596642Z","shell.execute_reply.started":"2022-07-09T09:33:35.589139Z","shell.execute_reply":"2022-07-09T09:33:35.595646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Creating a correlation matrix to see if any features are correlated**","metadata":{}},{"cell_type":"code","source":"fig,ax=plt.subplots(figsize=(10,10))\ncorrmat = df.corr()\ntop_corr_features = corrmat.index\nsns.heatmap(df[top_corr_features].corr(),annot=True,cmap='viridis',ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:35.598020Z","iopub.execute_input":"2022-07-09T09:33:35.598285Z","iopub.status.idle":"2022-07-09T09:33:36.029404Z","shell.execute_reply.started":"2022-07-09T09:33:35.598259Z","shell.execute_reply":"2022-07-09T09:33:36.028751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above image we can see none of the features are highly correlated with any other features where correlation excedes 0.8 or 0.9.So no value needs to be dropped in the dataset.","metadata":{}},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.030914Z","iopub.execute_input":"2022-07-09T09:33:36.031692Z","iopub.status.idle":"2022-07-09T09:33:36.058536Z","shell.execute_reply.started":"2022-07-09T09:33:36.031650Z","shell.execute_reply":"2022-07-09T09:33:36.057558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Pre-processing.","metadata":{}},{"cell_type":"markdown","source":" We will remove some columns from both the datasets which we are not going to use .","metadata":{}},{"cell_type":"code","source":"# Scving passenger id for submission\nPass_ID = test['PassengerId']\ndf.drop(columns=['PassengerId','Cabin','Name'],axis=1,inplace=True)\ntest.drop(columns=['PassengerId','Cabin','Name'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.060771Z","iopub.execute_input":"2022-07-09T09:33:36.061334Z","iopub.status.idle":"2022-07-09T09:33:36.073851Z","shell.execute_reply.started":"2022-07-09T09:33:36.061303Z","shell.execute_reply":"2022-07-09T09:33:36.072511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(colored(\"The null values in training data set are :\",'red',attrs=['bold','underline']))\nprint(colored(df.isna().sum(),'grey',attrs=['bold']))\nprint(colored(f'Total missing values : {df.isna().sum().sum()}','blue',attrs=['bold','underline']))\nprint(colored(\"The null values in test data set are :\",'red',attrs=['bold','underline']))\nprint(colored(test.isna().sum(),'grey',attrs=['bold']))\nprint(colored(f'Total missing values : {test.isna().sum().sum()}','blue',attrs=['bold','underline']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.075874Z","iopub.execute_input":"2022-07-09T09:33:36.076176Z","iopub.status.idle":"2022-07-09T09:33:36.097426Z","shell.execute_reply.started":"2022-07-09T09:33:36.076147Z","shell.execute_reply":"2022-07-09T09:33:36.096844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Handling the missing values**","metadata":{}},{"cell_type":"code","source":"columns = test.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.098364Z","iopub.execute_input":"2022-07-09T09:33:36.099045Z","iopub.status.idle":"2022-07-09T09:33:36.103728Z","shell.execute_reply.started":"2022-07-09T09:33:36.099015Z","shell.execute_reply":"2022-07-09T09:33:36.102520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.104977Z","iopub.execute_input":"2022-07-09T09:33:36.105255Z","iopub.status.idle":"2022-07-09T09:33:36.116303Z","shell.execute_reply.started":"2022-07-09T09:33:36.105226Z","shell.execute_reply":"2022-07-09T09:33:36.115151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in columns:\n    if col in ['Age','RoomService','FoodCourt','ShoppingMall','Spa','VRDeck',]:\n        df[col].fillna(df[col].median(),inplace=True)\n        test[col].fillna(test[col].median(),inplace=True)\n    else:\n        df[col].fillna(df[col].mode()[0],inplace=True)\n        test[col].fillna(test[col].mode()[0],inplace=True)\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.118012Z","iopub.execute_input":"2022-07-09T09:33:36.118646Z","iopub.status.idle":"2022-07-09T09:33:36.143336Z","shell.execute_reply.started":"2022-07-09T09:33:36.118618Z","shell.execute_reply":"2022-07-09T09:33:36.142612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(colored('Null Values after : ','green',attrs=['bold','underline']))\nprint(colored(f'For training data : {df.isna().sum().sum()}','grey',attrs=['bold']))\nprint(colored(f'For test data : {test.isna().sum().sum()}','grey',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.144514Z","iopub.execute_input":"2022-07-09T09:33:36.144726Z","iopub.status.idle":"2022-07-09T09:33:36.154901Z","shell.execute_reply.started":"2022-07-09T09:33:36.144706Z","shell.execute_reply":"2022-07-09T09:33:36.153890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Encoding labels and Scaling the data**","metadata":{}},{"cell_type":"markdown","source":" - Objects will be LabelEncoded\n - Bools will be converted to int\n - Floats will be scaled usinf MinMaxScaler","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler,LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.156520Z","iopub.execute_input":"2022-07-09T09:33:36.156782Z","iopub.status.idle":"2022-07-09T09:33:36.223443Z","shell.execute_reply.started":"2022-07-09T09:33:36.156759Z","shell.execute_reply":"2022-07-09T09:33:36.222629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in columns:\n    if df[col].dtype == 'O':\n        encoder = LabelEncoder()\n        df[col] = encoder.fit_transform(df[col])\n        test[col] = encoder.fit_transform(test[col])\n        \n    elif df[col].dtype == 'bool':\n        df[col] = df[col].astype('int')\n        test[col] = test[col].astype('int')\n\n# Turning the transported column into int\ndf['Transported'] = df['Transported'].astype('int')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.224532Z","iopub.execute_input":"2022-07-09T09:33:36.224803Z","iopub.status.idle":"2022-07-09T09:33:36.241352Z","shell.execute_reply.started":"2022-07-09T09:33:36.224775Z","shell.execute_reply":"2022-07-09T09:33:36.240813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.242653Z","iopub.execute_input":"2022-07-09T09:33:36.242920Z","iopub.status.idle":"2022-07-09T09:33:36.258395Z","shell.execute_reply.started":"2022-07-09T09:33:36.242893Z","shell.execute_reply":"2022-07-09T09:33:36.257480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scaling Age column\nscaler=MinMaxScaler()\n# Dimension did not match so create a varianlt with 2 dimension and then pass it to df after wards\nscale_col = ['Age',]\ndf['Age'] = scaler.fit_transform(df[scale_col])\ntest['Age'] = scaler.fit_transform(test[scale_col])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.259812Z","iopub.execute_input":"2022-07-09T09:33:36.260778Z","iopub.status.idle":"2022-07-09T09:33:36.275282Z","shell.execute_reply.started":"2022-07-09T09:33:36.260740Z","shell.execute_reply":"2022-07-09T09:33:36.274125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, Y = df.drop(columns=['Transported'],axis=1),df[['Transported']]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.276136Z","iopub.execute_input":"2022-07-09T09:33:36.276839Z","iopub.status.idle":"2022-07-09T09:33:36.282046Z","shell.execute_reply.started":"2022-07-09T09:33:36.276816Z","shell.execute_reply":"2022-07-09T09:33:36.281460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.283059Z","iopub.execute_input":"2022-07-09T09:33:36.283460Z","iopub.status.idle":"2022-07-09T09:33:36.297669Z","shell.execute_reply.started":"2022-07-09T09:33:36.283414Z","shell.execute_reply":"2022-07-09T09:33:36.296687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train Test Split**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.298929Z","iopub.execute_input":"2022-07-09T09:33:36.299601Z","iopub.status.idle":"2022-07-09T09:33:36.355630Z","shell.execute_reply.started":"2022-07-09T09:33:36.299573Z","shell.execute_reply":"2022-07-09T09:33:36.354631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_val,Y_train,Y_val = train_test_split(X,Y,test_size=0.2,random_state=8,stratify=Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.357295Z","iopub.execute_input":"2022-07-09T09:33:36.357940Z","iopub.status.idle":"2022-07-09T09:33:36.417231Z","shell.execute_reply.started":"2022-07-09T09:33:36.357902Z","shell.execute_reply":"2022-07-09T09:33:36.416491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(colored(f'Rows and column in X_train : {X_train.shape}','red',attrs=['bold']))\nprint(colored(f'Rows and column in Y_train : {Y_train.shape}','red',attrs=['bold']))\nprint(colored(f'Rows and column in X_val : {X_val.shape}','cyan',attrs=['bold']))\nprint(colored(f'Rows and column in Y_val : {Y_val.shape}','cyan',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.418613Z","iopub.execute_input":"2022-07-09T09:33:36.419086Z","iopub.status.idle":"2022-07-09T09:33:36.424718Z","shell.execute_reply.started":"2022-07-09T09:33:36.419061Z","shell.execute_reply":"2022-07-09T09:33:36.423633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Machine Learning Models.","metadata":{}},{"cell_type":"markdown","source":"- Logistic Regression\n- Support Vector Classifier\n- Stochastic Gradient Descent Classifier\n- Random Forest Classifier\n- XGBoost Classifier\n- AdaBoost Classifier\n\n","metadata":{}},{"cell_type":"markdown","source":"**Now we will use the above models to find out the most accurate model.**","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier,AdaBoostClassifier\nfrom sklearn.linear_model import LogisticRegression,SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.svm import SVC","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.426133Z","iopub.execute_input":"2022-07-09T09:33:36.427203Z","iopub.status.idle":"2022-07-09T09:33:36.756119Z","shell.execute_reply.started":"2022-07-09T09:33:36.427166Z","shell.execute_reply":"2022-07-09T09:33:36.754946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dictionary to store model accuracy\nmodel_dict = {}","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.757395Z","iopub.execute_input":"2022-07-09T09:33:36.757987Z","iopub.status.idle":"2022-07-09T09:33:36.761780Z","shell.execute_reply.started":"2022-07-09T09:33:36.757964Z","shell.execute_reply":"2022-07-09T09:33:36.760812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For checking the accuracy of each model\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.762775Z","iopub.execute_input":"2022-07-09T09:33:36.763007Z","iopub.status.idle":"2022-07-09T09:33:36.773792Z","shell.execute_reply.started":"2022-07-09T09:33:36.762986Z","shell.execute_reply":"2022-07-09T09:33:36.772830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Logistic Regression","metadata":{}},{"cell_type":"code","source":"classifier = LogisticRegression()\nclassifier.fit(X_train,Y_train)\nY_pred = classifier.predict(X_val)\nacc_log_reg = accuracy_score(Y_val,Y_pred)\nmodel_dict['Logistic Regression'] = acc_log_reg\nprint(colored(acc_log_reg,'magenta',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.774737Z","iopub.execute_input":"2022-07-09T09:33:36.774941Z","iopub.status.idle":"2022-07-09T09:33:36.863143Z","shell.execute_reply.started":"2022-07-09T09:33:36.774922Z","shell.execute_reply":"2022-07-09T09:33:36.862483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Supported Vector Machine","metadata":{}},{"cell_type":"code","source":"classifier = SVC(random_state=2)\nclassifier.fit(X_train,Y_train)\nY_pred = classifier.predict(X_val)\nacc_supp_vec = accuracy_score(Y_val,Y_pred)\nmodel_dict['Supported Vector Machine'] = acc_supp_vec\nprint(colored(acc_supp_vec,'magenta',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:36.864083Z","iopub.execute_input":"2022-07-09T09:33:36.864503Z","iopub.status.idle":"2022-07-09T09:33:38.620979Z","shell.execute_reply.started":"2022-07-09T09:33:36.864476Z","shell.execute_reply":"2022-07-09T09:33:38.619836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Stochastic Gradient Descent","metadata":{}},{"cell_type":"code","source":"classifier = SGDClassifier(random_state=10)\nclassifier.fit(X_train,Y_train)\nY_pred = classifier.predict(X_val)\nacc_sto_gra = accuracy_score(Y_val,Y_pred)\nmodel_dict['Stochastic Gradient Descent'] = acc_sto_gra\nprint(colored(acc_sto_gra,'magenta',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:38.622571Z","iopub.execute_input":"2022-07-09T09:33:38.622934Z","iopub.status.idle":"2022-07-09T09:33:38.657648Z","shell.execute_reply.started":"2022-07-09T09:33:38.622903Z","shell.execute_reply":"2022-07-09T09:33:38.656758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"classifier = RandomForestClassifier(random_state=40)\nclassifier.fit(X_train,Y_train)\nY_pred = classifier.predict(X_val)\nacc_rand_for = accuracy_score(Y_val,Y_pred)\nmodel_dict['Random Forest Classifier'] = acc_rand_for\nprint(colored(acc_rand_for,'magenta',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:38.658825Z","iopub.execute_input":"2022-07-09T09:33:38.659081Z","iopub.status.idle":"2022-07-09T09:33:39.567953Z","shell.execute_reply.started":"2022-07-09T09:33:38.659055Z","shell.execute_reply":"2022-07-09T09:33:39.566818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier = XGBClassifier(random_state=20,eval_metric='rmse')\nclassifier.fit(X_train,Y_train)\nY_pred = classifier.predict(X_val)\nacc_XG_boo = accuracy_score(Y_val,Y_pred)\nmodel_dict['Extreme Gradient Boost'] = acc_XG_boo\nprint(colored(acc_XG_boo,'magenta',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:39.569234Z","iopub.execute_input":"2022-07-09T09:33:39.569486Z","iopub.status.idle":"2022-07-09T09:33:40.169900Z","shell.execute_reply.started":"2022-07-09T09:33:39.569464Z","shell.execute_reply":"2022-07-09T09:33:40.169143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Adaptive Boosting(ADaBoost)","metadata":{}},{"cell_type":"code","source":"dtc = DecisionTreeClassifier(random_state=40)\nclassifier = AdaBoostClassifier(dtc,random_state=42)\nclassifier.fit(X_train,Y_train)\nY_pred = classifier.predict(X_val)\nacc_Ada_boo = accuracy_score(Y_val,Y_pred)\nmodel_dict['Adaptive Boosting'] = acc_Ada_boo\nprint(colored(acc_Ada_boo,'magenta',attrs=['bold']))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:40.171242Z","iopub.execute_input":"2022-07-09T09:33:40.171774Z","iopub.status.idle":"2022-07-09T09:33:41.257841Z","shell.execute_reply.started":"2022-07-09T09:33:40.171742Z","shell.execute_reply":"2022-07-09T09:33:41.256786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_dict","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:41.258994Z","iopub.execute_input":"2022-07-09T09:33:41.259315Z","iopub.status.idle":"2022-07-09T09:33:41.266774Z","shell.execute_reply.started":"2022-07-09T09:33:41.259283Z","shell.execute_reply":"2022-07-09T09:33:41.265678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_df = pd.DataFrame(model_dict,index=['Accuracy']).T","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:41.268647Z","iopub.execute_input":"2022-07-09T09:33:41.269569Z","iopub.status.idle":"2022-07-09T09:33:41.279332Z","shell.execute_reply.started":"2022-07-09T09:33:41.269530Z","shell.execute_reply":"2022-07-09T09:33:41.278401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These are the respective model with their accuracies.","metadata":{}},{"cell_type":"code","source":"model_df","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:41.281101Z","iopub.execute_input":"2022-07-09T09:33:41.281759Z","iopub.status.idle":"2022-07-09T09:33:41.293976Z","shell.execute_reply.started":"2022-07-09T09:33:41.281707Z","shell.execute_reply":"2022-07-09T09:33:41.293145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**So the models with the highest accuracies are Logistic Regression,SVM,XGB**","metadata":{}},{"cell_type":"markdown","source":"Now we will perform Grid Search on these 3 models.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:41.295067Z","iopub.execute_input":"2022-07-09T09:33:41.295395Z","iopub.status.idle":"2022-07-09T09:33:41.298707Z","shell.execute_reply.started":"2022-07-09T09:33:41.295373Z","shell.execute_reply":"2022-07-09T09:33:41.297855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_of_models = [LogisticRegression(max_iter=1000),SVC(max_iter=1000),XGBClassifier(eval_metric='rmse')]\nmodel_hyperparameters = {\n    'log_reg': {\n        'C' : [1,5,10,15,20]\n    },\n    'SVC' : {\n        'kernel' : ['linear','poly','rbf','sigmoid'],\n        'C' : [1,5,10,15,20]\n    },\n    'XG_Boost' : {\n        'learning_rate': [0.1,0.125, 0.075],\n        'n_estimators': [15, 50, 100],\n        'max_depth': [3, 4, 5, 6]\n    }\n    \n}","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:41.299571Z","iopub.execute_input":"2022-07-09T09:33:41.299943Z","iopub.status.idle":"2022-07-09T09:33:41.309627Z","shell.execute_reply.started":"2022-07-09T09:33:41.299922Z","shell.execute_reply":"2022-07-09T09:33:41.308880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keys = list(model_hyperparameters.keys())\nkeys","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:41.310966Z","iopub.execute_input":"2022-07-09T09:33:41.311207Z","iopub.status.idle":"2022-07-09T09:33:41.325376Z","shell.execute_reply.started":"2022-07-09T09:33:41.311180Z","shell.execute_reply":"2022-07-09T09:33:41.324505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"global_result = []\n# Function for model selection\ndef model_selection(model_list,hyperparameter_list):\n    i=0\n    for model in model_list:\n        model_name = keys[i]\n        params = model_hyperparameters[model_name]\n        i+=1\n        print(colored(model,'blue',attrs=['bold','underline']))\n        print(colored(params,'magenta',attrs=['bold']))\n        \n        classifier = GridSearchCV(model,params,cv=5)\n        classifier.fit(X,Y)\n        \n        global_result.append({\n            'Model Used' : [model],\n            'Highest Score' : [classifier.best_score_],\n            'Best Parameters' : [classifier.best_params_]\n        })\n    \n    print(colored(global_result,'red',attrs=['bold']))\n    result_dataframe = pd.DataFrame(global_result,columns=['Model Used','Highest Score','Best Parameters'])\n    return result_dataframe","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:41.326919Z","iopub.execute_input":"2022-07-09T09:33:41.327286Z","iopub.status.idle":"2022-07-09T09:33:41.338373Z","shell.execute_reply.started":"2022-07-09T09:33:41.327212Z","shell.execute_reply":"2022-07-09T09:33:41.337186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_selection(list_of_models,model_hyperparameters)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:33:41.339861Z","iopub.execute_input":"2022-07-09T09:33:41.340349Z","iopub.status.idle":"2022-07-09T09:35:57.571257Z","shell.execute_reply.started":"2022-07-09T09:33:41.340321Z","shell.execute_reply":"2022-07-09T09:35:57.570205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**So we can see from the above Dataframe that XGBClassifier has the highest accuracy of all the model with its respective parameters**","metadata":{}},{"cell_type":"code","source":"classifier=XGBClassifier(learning_rate=0.125,n_estimators=50,eval_metric='rmse')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:35:57.572201Z","iopub.execute_input":"2022-07-09T09:35:57.572423Z","iopub.status.idle":"2022-07-09T09:35:57.577262Z","shell.execute_reply.started":"2022-07-09T09:35:57.572402Z","shell.execute_reply":"2022-07-09T09:35:57.576256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:35:57.578593Z","iopub.execute_input":"2022-07-09T09:35:57.579145Z","iopub.status.idle":"2022-07-09T09:35:57.600609Z","shell.execute_reply.started":"2022-07-09T09:35:57.579117Z","shell.execute_reply":"2022-07-09T09:35:57.599349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# have to call fit for some reason for XGBoost or it shows error.\nclassifier.fit(X_val,Y_val)\nsubmission = classifier.predict(test)\ndata = {\n    'PassengerId' : Pass_ID,\n    'Transported' : np.asarray(submission).astype('bool')\n}\nsubmission_df = pd.DataFrame(data).reset_index()\n\nsubmission_df.drop(columns=['index'],inplace=True)\nos.chdir('/kaggle/working/')\nsubmission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:35:57.602022Z","iopub.execute_input":"2022-07-09T09:35:57.602320Z","iopub.status.idle":"2022-07-09T09:35:57.805079Z","shell.execute_reply.started":"2022-07-09T09:35:57.602293Z","shell.execute_reply":"2022-07-09T09:35:57.803155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}