{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# First of all to build a model we will need to import some required libraries.\n> we will be analizing and visualizing the data with the help of graphs as well.\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cufflinks as cf\ncf.go_offline()\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:24.317997Z","iopub.execute_input":"2022-08-06T11:51:24.319087Z","iopub.status.idle":"2022-08-06T11:51:24.327294Z","shell.execute_reply.started":"2022-08-06T11:51:24.319033Z","shell.execute_reply":"2022-08-06T11:51:24.326423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reding and visualizing the data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/titanic/train.csv')\ntest = pd.read_csv('../input/titanic/test.csv')\nsubmission = pd.read_csv('../input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:24.801031Z","iopub.execute_input":"2022-08-06T11:51:24.802128Z","iopub.status.idle":"2022-08-06T11:51:24.823570Z","shell.execute_reply.started":"2022-08-06T11:51:24.802090Z","shell.execute_reply":"2022-08-06T11:51:24.822570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()  # by default head gives us first 5 rows from our dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:25.181477Z","iopub.execute_input":"2022-08-06T11:51:25.182227Z","iopub.status.idle":"2022-08-06T11:51:25.199238Z","shell.execute_reply.started":"2022-08-06T11:51:25.182188Z","shell.execute_reply":"2022-08-06T11:51:25.198300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:25.429399Z","iopub.execute_input":"2022-08-06T11:51:25.429811Z","iopub.status.idle":"2022-08-06T11:51:25.447734Z","shell.execute_reply.started":"2022-08-06T11:51:25.429775Z","shell.execute_reply":"2022-08-06T11:51:25.446798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking total missing values in our data\nlen(train.isna())","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:25.732930Z","iopub.execute_input":"2022-08-06T11:51:25.734082Z","iopub.status.idle":"2022-08-06T11:51:25.741939Z","shell.execute_reply.started":"2022-08-06T11:51:25.734040Z","shell.execute_reply":"2022-08-06T11:51:25.740939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing missing values through the graph\nimport missingno\nmissingno.matrix(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:26.005338Z","iopub.execute_input":"2022-08-06T11:51:26.006512Z","iopub.status.idle":"2022-08-06T11:51:26.584233Z","shell.execute_reply.started":"2022-08-06T11:51:26.006467Z","shell.execute_reply":"2022-08-06T11:51:26.583001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> From the above graph we clearly see that there are a lot of missing values in age and cabin columns. And we will try to fix those ahead.","metadata":{}},{"cell_type":"code","source":"train['Survived'].iplot(kind='hist')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:26.744898Z","iopub.execute_input":"2022-08-06T11:51:26.745309Z","iopub.status.idle":"2022-08-06T11:51:26.810028Z","shell.execute_reply.started":"2022-08-06T11:51:26.745276Z","shell.execute_reply":"2022-08-06T11:51:26.809116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking number of male and female passingers \ntrain.pivot(columns='Sex', values='Sex').iplot(kind='hist', title = 'Gender count', yTitle ='Number of people', xTitle='Gender')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:27.153161Z","iopub.execute_input":"2022-08-06T11:51:27.153864Z","iopub.status.idle":"2022-08-06T11:51:27.232017Z","shell.execute_reply.started":"2022-08-06T11:51:27.153820Z","shell.execute_reply":"2022-08-06T11:51:27.230582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking Fare for each pasinger calss\ntrain.pivot(columns='Pclass', values='Fare').iplot(kind='box', xTitle= 'Passenger Class',yTitle='Fare')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:27.605075Z","iopub.execute_input":"2022-08-06T11:51:27.605461Z","iopub.status.idle":"2022-08-06T11:51:27.700453Z","shell.execute_reply.started":"2022-08-06T11:51:27.605430Z","shell.execute_reply":"2022-08-06T11:51:27.699187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# chekcing age per class\ntrain.pivot(columns='Pclass', values='Age').iplot(kind='box', xTitle= 'Passenger Class',yTitle='Age', title = \"Age according to passenger class\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:28.105048Z","iopub.execute_input":"2022-08-06T11:51:28.106030Z","iopub.status.idle":"2022-08-06T11:51:28.196081Z","shell.execute_reply.started":"2022-08-06T11:51:28.105990Z","shell.execute_reply":"2022-08-06T11:51:28.195004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the age of most and if it forms bells curve\ntrain['Age'].dropna().iplot(kind='hist', bins=30, xTitle = \"Age\", yTitle='Number of People',title ='Age Graph')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:28.864890Z","iopub.execute_input":"2022-08-06T11:51:28.866011Z","iopub.status.idle":"2022-08-06T11:51:28.926384Z","shell.execute_reply.started":"2022-08-06T11:51:28.865950Z","shell.execute_reply":"2022-08-06T11:51:28.925505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# survived and dead by gender\nsns.catplot(x =\"Sex\", hue =\"Survived\",kind =\"count\", data = train)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:29.629091Z","iopub.execute_input":"2022-08-06T11:51:29.630128Z","iopub.status.idle":"2022-08-06T11:51:29.872967Z","shell.execute_reply.started":"2022-08-06T11:51:29.630087Z","shell.execute_reply":"2022-08-06T11:51:29.871443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking survival rate by embraked and passener\nsns.catplot(x ='Embarked', hue ='Survived',kind ='count', col ='Pclass', data = train)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:30.653224Z","iopub.execute_input":"2022-08-06T11:51:30.653634Z","iopub.status.idle":"2022-08-06T11:51:31.307364Z","shell.execute_reply.started":"2022-08-06T11:51:30.653600Z","shell.execute_reply":"2022-08-06T11:51:31.306156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pclass vs Survivd graphs\nsns.catplot(x ='Sex', hue ='Survived',kind ='count', col ='Pclass', data = train)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:51:40.232760Z","iopub.execute_input":"2022-08-06T11:51:40.233246Z","iopub.status.idle":"2022-08-06T11:51:40.764674Z","shell.execute_reply.started":"2022-08-06T11:51:40.233208Z","shell.execute_reply":"2022-08-06T11:51:40.763346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defining a function to fill all the missing age values \ndef fill_age(cols):\n    Age=cols[0]\n    Pclass=cols[1]\n    if pd.isnull(Age): \n        if Pclass == 1:\n            return 37 # 37 is avg age as it can be seen in box plot for pasinger in Pclass 1\n        if Pclass == 2:\n            return 29 # 29 is avg age as it can be seen in box plot for pasinger in Pclass 1\n        if Pclass == 3:\n            return 24 # 24 is avg age as it can be seen in box plot for pasinger in Pclass 1\n    else:\n        return Age","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.622140Z","iopub.execute_input":"2022-08-06T11:19:18.622689Z","iopub.status.idle":"2022-08-06T11:19:18.630441Z","shell.execute_reply.started":"2022-08-06T11:19:18.622643Z","shell.execute_reply":"2022-08-06T11:19:18.628969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Applying the above funcation in our train and test data sets\ntrain['Age'] = train[['Pclass','Age']].apply(fill_age,axis=1)\ntest['Age'] = test[['Pclass','Age']].apply(fill_age,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.634419Z","iopub.execute_input":"2022-08-06T11:19:18.634923Z","iopub.status.idle":"2022-08-06T11:19:18.660440Z","shell.execute_reply.started":"2022-08-06T11:19:18.634891Z","shell.execute_reply":"2022-08-06T11:19:18.659581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One hot encoding sex column (converting it into 0 and 1)\nsex1 = pd.get_dummies(train['Sex'],drop_first=True)\nsex2 = pd.get_dummies(test['Sex'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.661845Z","iopub.execute_input":"2022-08-06T11:19:18.662201Z","iopub.status.idle":"2022-08-06T11:19:18.669918Z","shell.execute_reply.started":"2022-08-06T11:19:18.662168Z","shell.execute_reply":"2022-08-06T11:19:18.668742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# similarly one hot encoding embarked data\nembarked1 = pd.get_dummies(train['Embarked'],drop_first=True)\nembarked2 = pd.get_dummies(test['Embarked'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.671497Z","iopub.execute_input":"2022-08-06T11:19:18.671944Z","iopub.status.idle":"2022-08-06T11:19:18.682352Z","shell.execute_reply.started":"2022-08-06T11:19:18.671901Z","shell.execute_reply":"2022-08-06T11:19:18.681222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One hot encodung parch column as well\nparch1 = pd.get_dummies(train['Parch'],drop_first=True)\nparch2 = pd.get_dummies(test['Parch'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.683926Z","iopub.execute_input":"2022-08-06T11:19:18.685127Z","iopub.status.idle":"2022-08-06T11:19:18.692968Z","shell.execute_reply.started":"2022-08-06T11:19:18.685089Z","shell.execute_reply":"2022-08-06T11:19:18.692025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now we need to drop Unwanted data and add One Hot encoded data","metadata":{}},{"cell_type":"code","source":"train.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.694152Z","iopub.execute_input":"2022-08-06T11:19:18.695280Z","iopub.status.idle":"2022-08-06T11:19:18.714771Z","shell.execute_reply.started":"2022-08-06T11:19:18.695230Z","shell.execute_reply":"2022-08-06T11:19:18.713937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dropping columns\ntrain.drop(['PassengerId','Name','Sex','Ticket','Cabin','Embarked'],inplace =True, axis=1)\ntest.drop(['PassengerId','Name','Sex','Ticket','Cabin','Embarked'],inplace =True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.716330Z","iopub.execute_input":"2022-08-06T11:19:18.716687Z","iopub.status.idle":"2022-08-06T11:19:18.724724Z","shell.execute_reply.started":"2022-08-06T11:19:18.716655Z","shell.execute_reply":"2022-08-06T11:19:18.723534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adding one hot encoded values\ntrain = pd.concat([train,sex1,embarked1,parch1],axis=1)\ntest = pd.concat([test,sex2,embarked2,parch2],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.726303Z","iopub.execute_input":"2022-08-06T11:19:18.726725Z","iopub.status.idle":"2022-08-06T11:19:18.739867Z","shell.execute_reply.started":"2022-08-06T11:19:18.726684Z","shell.execute_reply":"2022-08-06T11:19:18.738563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop([9],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.741248Z","iopub.execute_input":"2022-08-06T11:19:18.741583Z","iopub.status.idle":"2022-08-06T11:19:18.754578Z","shell.execute_reply.started":"2022-08-06T11:19:18.741554Z","shell.execute_reply":"2022-08-06T11:19:18.753687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalizing the values \ntrain['Age']=np.log(train['Age'])\ntest['Age']=np.log(test['Age'])","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:18.755815Z","iopub.execute_input":"2022-08-06T11:19:18.756364Z","iopub.status.idle":"2022-08-06T11:19:19.048734Z","shell.execute_reply.started":"2022-08-06T11:19:18.756330Z","shell.execute_reply":"2022-08-06T11:19:19.047833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:19.049962Z","iopub.execute_input":"2022-08-06T11:19:19.050539Z","iopub.status.idle":"2022-08-06T11:19:19.073008Z","shell.execute_reply.started":"2022-08-06T11:19:19.050504Z","shell.execute_reply":"2022-08-06T11:19:19.071726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checnking again if we have missing values \nmissingno.matrix(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:19.074256Z","iopub.execute_input":"2022-08-06T11:19:19.074616Z","iopub.status.idle":"2022-08-06T11:19:19.592885Z","shell.execute_reply.started":"2022-08-06T11:19:19.074585Z","shell.execute_reply":"2022-08-06T11:19:19.591569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating training and testing data sets","metadata":{}},{"cell_type":"code","source":"x = train.drop(['Survived'], axis=1)\ny = train['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:19.594554Z","iopub.execute_input":"2022-08-06T11:19:19.595053Z","iopub.status.idle":"2022-08-06T11:19:19.602913Z","shell.execute_reply.started":"2022-08-06T11:19:19.595007Z","shell.execute_reply":"2022-08-06T11:19:19.601817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_test, y_train, y_test = train_test_split(x,y, test_size=0.2,random_state=4)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:19.604430Z","iopub.execute_input":"2022-08-06T11:19:19.605165Z","iopub.status.idle":"2022-08-06T11:19:19.615055Z","shell.execute_reply.started":"2022-08-06T11:19:19.605130Z","shell.execute_reply":"2022-08-06T11:19:19.614034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating model:\n> We will use the XGBoost for the classification. We don’t know the best parameters for our model, so we will let GridSearchCV find them for us. We will just define the parameter range for GridSearchCV to search within.","metadata":{}},{"cell_type":"code","source":" #Defining the parameters to search within\nparam_grid = {\n'n_estimators': range(6, 10),\n'max_depth': range(3, 8),\n'learning_rate': [.2, .3, .4],\n'colsample_bytree': [.7, .8, .9, 1]\n}\n#Specifying our classifier\nxgb = XGBClassifier()\n#Searching for the best parameters\ng_search = GridSearchCV(estimator = xgb, param_grid = param_grid,\ncv = 3, n_jobs = 1, verbose = 0, return_train_score=True)\n#Fitting the model using best parameters found\ng_search.fit(x_train, y_train)\n#Printing the best parameters found\nprint(g_search.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:19.616453Z","iopub.execute_input":"2022-08-06T11:19:19.617177Z","iopub.status.idle":"2022-08-06T11:19:57.108137Z","shell.execute_reply.started":"2022-08-06T11:19:19.617136Z","shell.execute_reply":"2022-08-06T11:19:57.106993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Calculating the model's score\ng_search.score(x_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:19:57.112614Z","iopub.execute_input":"2022-08-06T11:19:57.114951Z","iopub.status.idle":"2022-08-06T11:19:57.131197Z","shell.execute_reply.started":"2022-08-06T11:19:57.114904Z","shell.execute_reply":"2022-08-06T11:19:57.130290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predicting and Creating Sample_submission","metadata":{}},{"cell_type":"code","source":"pred = g_search.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:21:50.160569Z","iopub.execute_input":"2022-08-06T11:21:50.160940Z","iopub.status.idle":"2022-08-06T11:21:50.171591Z","shell.execute_reply.started":"2022-08-06T11:21:50.160909Z","shell.execute_reply":"2022-08-06T11:21:50.170620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:23:48.848890Z","iopub.execute_input":"2022-08-06T11:23:48.849290Z","iopub.status.idle":"2022-08-06T11:23:48.860631Z","shell.execute_reply.started":"2022-08-06T11:23:48.849258Z","shell.execute_reply":"2022-08-06T11:23:48.859632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = pd.Series(pred, name='Survived')\nsubmission['Survived'] = pred\nsubmission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:26:59.520841Z","iopub.execute_input":"2022-08-06T11:26:59.521341Z","iopub.status.idle":"2022-08-06T11:26:59.531878Z","shell.execute_reply.started":"2022-08-06T11:26:59.521302Z","shell.execute_reply":"2022-08-06T11:26:59.530493Z"},"trusted":true},"execution_count":null,"outputs":[]}]}