{"cells":[{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#loading all the required packages\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4cb442278996adf188dfa76f114522041d3d564"},"cell_type":"code","source":"#loading the train and test dataset\ntrain = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\ntrain_raw = train.copy()\ntest_raw = test.copy()\nprint(\"Train : {}\".format(train.shape))\nprint(\"Test : {}\".format(test.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b77d5e89b26d63ce2f28c75a6ce3d9a745016722"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b106c847c356ad865f2964c3585633b7394c5f67"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4379f9f765209f880c979443097e0089dbe39451"},"cell_type":"code","source":"#some useful information about the data\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"882f4ccee4c38d87ade6358330d23253df0eabd7"},"cell_type":"markdown","source":"* There are 7 features with numeric data and 5 features with categorical data.\n* There are missing values in 'Age','Cabin' and 'Embarked' columns.\n"},{"metadata":{"trusted":true,"_uuid":"67e547dde22bd287515587e9d039891ff2655315"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"612c2a7c49838addfabac438da36378d8edb805e"},"cell_type":"markdown","source":"* There are 6 features with numeric data (No survival column) and 5 features with categorical data.\n* There are missing values in 'Age','Cabin' and 'Fare' columns."},{"metadata":{"_uuid":"e87ad3d208a24909049f4130a690a6ea31e76bdf"},"cell_type":"markdown","source":"**PassengerId**"},{"metadata":{"trusted":true,"_uuid":"653e01437188c954b8d95ad3604e29345785e494"},"cell_type":"code","source":"#dropping PassengerId columns from both train and test datasets\ntrain.drop(columns = ['PassengerId'],inplace = True)\ntest.drop(columns = ['PassengerId'],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"69dfd330d4ce196c266f97cfdc1ad418a582c6b5"},"cell_type":"markdown","source":"**Pclass**\n\nThese are not really numbers but they represent the Lower, Middle and Upper class. Hence converting this feature to object type"},{"metadata":{"trusted":true,"_uuid":"6d9a0f8eb771620f30546c8e9cea7f0810858fe4"},"cell_type":"code","source":"train['Pclass'] = train['Pclass'].astype(str)\ntest['Pclass'] = test['Pclass'].astype(str)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"15a8102c1d965298ff27bdec3bc01458cf43525f"},"cell_type":"markdown","source":"**Name** \n\nThe name of the passenger obviously won't help in predicting the survival but we can extract some features that do help in predicting the survival such as the *Title*\n\nHence extracting the Title and dropping the Name column"},{"metadata":{"trusted":true,"_uuid":"fee1f2bae24bfa57c906ecd9f07f8ac496b6c5d2"},"cell_type":"code","source":"for dataset in (train,test) :\n    dataset['Title'] = dataset['Name'].str.extract('([A-Za-z]+)\\.',expand = False)\n    #dropping Name column\n    dataset.drop(columns = ['Name'],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a23974bbd8152ba7020604e0bbea12d24ec73083"},"cell_type":"code","source":"train['Title'].groupby(by = train['Title']).count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be776c259cf134db9afc210daad94c4f85cae9cd"},"cell_type":"code","source":"test['Title'].groupby(by = test['Title']).count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9197a3a99a2178d358ec67f84a7caf2d2ef863b8"},"cell_type":"code","source":"#aggregating titles\nTitle_Dict = {\n                    \"Capt\":       \"Officer\",\n                    \"Col\":        \"Officer\",\n                    \"Major\":      \"Officer\",\n                    \"Jonkheer\":   \"Royalty\",\n                    \"Don\":        \"Royalty\",\n                    \"Sir\" :       \"Royalty\",\n                    \"Dr\":         \"Officer\",\n                    \"Rev\":        \"Officer\",\n                    \"Countess\": \"Royalty\",\n                    \"Dona\":       \"Royalty\",\n                    \"Mme\":        \"Mrs\",\n                    \"Mlle\":       \"Miss\",\n                    \"Ms\":         \"Mrs\",\n                    \"Mr\" :        \"Mr\",\n                    \"Mrs\" :       \"Mrs\",\n                    \"Miss\" :      \"Miss\",\n                    \"Master\" :    \"Master\",\n                    \"Lady\" :      \"Royalty\"\n\n                    }\ntrain['Title'] = train['Title'].map(Title_Dict)\ntest['Title'] = test['Title'].map(Title_Dict)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"201fd263972e3d3385bcf2799dcabba63992d2ef"},"cell_type":"markdown","source":"**Age**\n\n* Filling in the missing values of Age by the mean values of their respective Title\n* Then creating Age bins"},{"metadata":{"trusted":true,"_uuid":"2786a17e85750be5602cf13f03376f9ebb4c181b"},"cell_type":"code","source":"Age_dict = train.groupby(by = 'Title')['Age'].mean().astype(int).to_dict()\nAge_dict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5faf2aee4b49ff1fe53e20003a077bf2b2911c59"},"cell_type":"code","source":"#filling the missing values\nfor dataset in (train,test):\n    nan_idx = dataset.loc[dataset['Age'].isnull()].index\n    dataset.loc[nan_idx,'Age'] = dataset.loc[nan_idx,'Title'].map(Age_dict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ed846b6624b20a42d69cf357c21236e77f4d5c6"},"cell_type":"code","source":"#Creating Age bins\n#Taking a look at the categories\n#quantile based discretization\npd.qcut(train['Age'],q = 5).head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf5c714acfbe5d92366f77c62869a00d6a2d8b95"},"cell_type":"code","source":"#Let's create bins based on the above categories\nbins = [0,20,26,32,38,80]\ntrain['Age'] = pd.cut(train['Age'],bins = bins,\n                      labels = ['Age_{}'.format(str(x)) for x in np.arange(1,6,1)])\ntest['Age'] = pd.cut(test['Age'],bins = bins,\n                    labels = ['Age_{}'.format(str(x)) for x in np.arange(1,6,1)])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0f2751c7d4ebf4334d7fc2468323962cb0cf1bdb"},"cell_type":"markdown","source":"**SibSp and Parch**\n\nLet's extract new features from SibSp and Parch and drop these two features."},{"metadata":{"trusted":true,"_uuid":"8bed18453e6cb10fcc51628aed9d5824246ed977"},"cell_type":"code","source":"for dataset in (train,test):\n    #Creating a feature called the Family size\n    dataset['FamilySize'] = dataset['SibSp'] + dataset['Parch'] + 1\n    # Create new feature IsAlone from FamilySize\n    dataset['IsAlone'] = (dataset['FamilySize'] == 1) * 1\n\ntrain.drop(columns = ['SibSp','Parch'],inplace = True)\ntest.drop(columns = ['SibSp','Parch'],inplace = True)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f14c92a25739b9b9248e4d3cbac8472e58a0db6e"},"cell_type":"markdown","source":"**Ticket**"},{"metadata":{"trusted":true,"_uuid":"af83cc287cbc126bbbe3bd5e5c8ef6c28a280842"},"cell_type":"code","source":"#dropping ticket feature as ticket number doesn't help in predicting survival\ntrain.drop(columns = ['Ticket'],inplace = True)\ntest.drop(columns = ['Ticket'],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a61a2dc1bae75a85ceab6bf05e8aa7277eebd971"},"cell_type":"markdown","source":"**Fare**\n\n* Filling a missing value of Fare in the test set by the mean value of the respective Pclass\n* Creating Fare bins"},{"metadata":{"trusted":true,"_uuid":"563c3c1c0fda243509daa576ea3282d877971f91"},"cell_type":"code","source":"test.loc[test['Fare'].isnull()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9e9e51da480a79c0dd03c6b5f3330b5fc1de708"},"cell_type":"code","source":"test['Fare'].fillna(test.loc[test['Pclass'] == '3','Fare'].mean(),inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c560f55756cb6701a82a538011ecd71207dd02db"},"cell_type":"code","source":"#Creating fare bins\n#Quantile cut\npd.qcut(train['Fare'],q = 4).head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d07f328dbd885fd445f580a424d1a1a9c372050f"},"cell_type":"code","source":"#Creating fare bins based on the above categories\nfare_bins = [-0.001,7.91,14.454,31,513]\ntrain['Fare'] = pd.cut(train['Fare'],bins = fare_bins,\n                       labels = ['Fare_{}'.format(str(x)) for x in np.arange(1,5,1)])\ntest['Fare'] = pd.cut(test['Fare'],bins = fare_bins,\n                     labels = ['Fare_{}'.format(str(x)) for x in np.arange(1,5,1)])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"08b7cb5b9280d177fa5a9d950d10cc740af7e115"},"cell_type":"markdown","source":"**Cabin**\n\n* Filling the missing values of Cabin with 'U' - Unknown\n* Then extracting the first letter in the Cabin number which might be helpful in predicting the survival."},{"metadata":{"trusted":true,"_uuid":"2d21e85da4335964af326998b5ef922bf2bffb2b"},"cell_type":"code","source":"for dataset in (train,test):\n    dataset['Cabin'].fillna('U',inplace = True)\n    dataset['Cabin'] = dataset['Cabin'].apply(lambda x : x[0])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cc95d00b4b213a062fd4ceaf1121b4eca151792b"},"cell_type":"markdown","source":"**Embarked**\n\nFilling the missing values of the Embarked column with the mode of that column"},{"metadata":{"trusted":true,"_uuid":"b68d6ec3b9e86084e264e585d31eded6dfd7308b"},"cell_type":"code","source":"train['Embarked'].fillna(train['Embarked'].mode()[0],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2ba0f7c678e5f2c5dd646d2118c664a5d646e5e"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c5d30a812d0cf4c6f3dd181df5e9ccf43147b90"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76fc5a76652a7834bc306640cff101d44623003e"},"cell_type":"code","source":"train = pd.get_dummies(train)\ntest = pd.get_dummies(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74a18efb53666b3eadc6b399df9c1155e19866ec"},"cell_type":"code","source":"#separating out target variables and predictor variables\ny_train = train['Survived']\n#dropping Cabin_T column also as it is not present in the test dataset\ntrain.drop(columns = ['Survived','Cabin_T'],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a149abf89701b7a40aeeb1e84dc65b8b9bb1395f"},"cell_type":"markdown","source":"Now we have our dataset ready for training. Let's train different classifiers on the training set and analyze the results."},{"metadata":{"trusted":true,"_uuid":"348ad8acb1e0e207c139eb9bd2bd0589e748fc27"},"cell_type":"markdown","source":"# **Modelling**"},{"metadata":{"_uuid":"2a42627c76e2829ec003a239dc82c42c1a350510"},"cell_type":"markdown","source":"# Logistic Regression"},{"metadata":{"trusted":true,"_uuid":"8c47f22aa9f89da15e5edcaf76651b339162a2e4"},"cell_type":"code","source":"#defining our error metric\nfrom sklearn.model_selection import StratifiedKFold,cross_val_score\ndef accuracy(model):\n    skfold = StratifiedKFold(n_splits = 5,shuffle = True,random_state = 66)\n    acc = cross_val_score(model,X = train,y = y_train,scoring = 'accuracy',cv = skfold)\n    return acc.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7081981a5819b78fc86fe1554bbced5dcf7bbfe2"},"cell_type":"code","source":"#finding C in LogisticRegression model\n#from sklearn.linear_model import LogisticRegression\n#for c in [0.001,0.01,0.1,1,10,100]:\n    #lr = LogisticRegression(penalty='l2',solver = 'lbfgs',C = c,random_state = 6,max_iter = 500)\n    #print(\"{} : {:.4f}\".format(c,accuracy(lr)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60f6f433190007c05e0dcacc978485df9eb038c8"},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlr = LogisticRegression(penalty='l2',solver = 'lbfgs',C = 0.1,random_state = 6,max_iter = 100)\nprint(\"Logistic Regression score : {:.4f}\".format(accuracy(lr)))\nlr.fit(train,y_train)\npred_lr = lr.predict(test)\nsubmission_lr = pd.DataFrame({\"PassengerId\" : test_raw[\"PassengerId\"], \"Survived\" : pred_lr })\nsubmission_lr.to_csv(\"logistic_regression\",index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91ace6ad51d72eb2583d33e81c0d75d31e47a3cb"},"cell_type":"markdown","source":"The above submission scored **0.77033** on public leaderboard"},{"metadata":{"trusted":true,"_uuid":"20da53fac30d22390c385c847f7f1b6ef094adc3"},"cell_type":"markdown","source":"# Ensembling"},{"metadata":{"_uuid":"4b7408ea84b0458ac9aa816a42f3c473a7d5500a"},"cell_type":"markdown","source":"# Random Forest Classifier"},{"metadata":{"trusted":true,"_uuid":"ec0ecba5f1f24e19a73fda6b6a90d09e4f29a6c2"},"cell_type":"code","source":"#cross validation to tune the parameters of random forest\n#from sklearn.model_selection import GridSearchCV\n#from sklearn.ensemble import RandomForestClassifier\n#parameter_grid = {'n_estimators' : [10,50,100,200,500],\n#                 'criterion' : ['entropy','gini'],\n#                 'max_features' : ['log2', 'sqrt','auto'],\n#                  'min_samples_leaf': [1,5,8],\n#                 'max_depth': [50,80,90,100,110],\n #                 'min_samples_split':[2,3,5]\n  #               }\n#skfold = StratifiedKFold(n_splits = 5,shuffle = True,random_state = 66)\n#grid_search = GridSearchCV(RandomForestClassifier(random_state = 666),param_grid = parameter_grid,\n #                          scoring = 'accuracy',n_jobs = -1,iid = False,cv = skfold,verbose = 2 )\n#grid_search.fit(train,y_train)\n#print(grid_search.best_params_)\n#print(grid_search.best_score_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2464252456584b284c08a6ec4258d7a3a5181ebc"},"cell_type":"code","source":"#Let's fit a random forest model\nfrom sklearn.ensemble import RandomForestClassifier\nrf = RandomForestClassifier(n_estimators = 50,max_depth = 90,criterion = 'entropy',\n                            max_features = 'log2',min_samples_leaf = 5,random_state = 55,\n                            min_samples_split = 2) #parameters estimated using cross validation\nprint(\"Random Forest score : {:.4f}\".format(accuracy(rf)))\nrf.fit(train,y_train)\npred_rf = rf.predict(test)\nsubmission_rf = pd.DataFrame({\"PassengerId\" : test_raw[\"PassengerId\"], \"Survived\" : pred_rf })\nsubmission_rf.to_csv(\"random_forest\",index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7af7f81874bc9f6a60d3fa634b8cdc3dad63e68"},"cell_type":"code","source":"#cross validation\n# from sklearn.model_selection import GridSearchCV\n# import xgboost as xgb\n# param_grid = {#'n_estimators' : [10,100,200,300,400,500,700,1000],\n#               #'max_depth' : [2,3,4,5],\n#                # 'min_child_weight' : [1,2,3,4]\n#                 #'gamma' : [0.0,0.1,0.2,0.3,0.4,0.5],\n#                 #'colsample_bytree' : [0.6,0.7,0.8,0.9,1.0],\n#                 #'subsample' : [0.6,0.7,0.8,0.9,1.0],\n#                 #'reg_alpha' : [0.01,0.03,0.1,0.3,1,3,10,30],\n#                 #'reg_lambda' :[0.01,0.03,0.1,0.3,1,3,10,30]\n#              } \n# skfold = StratifiedKFold(n_splits = 5,shuffle = True,random_state = 66)\n# XGB = xgb.XGBClassifier(learning_rate = 0.05,n_jobs = -1,max_depth = 2,n_estimators = 200,\n#                        subsample = 0.9,colsample_bytree = 0.9,min_child_weight = 1,\n#                         gamma = 0.0,reg_alpha = 0.01,reg_lambda = 1,random_state = 66)\n# grid_search = GridSearchCV(XGB,param_grid = param_grid,scoring = 'accuracy',\n#                            n_jobs = -1,iid = False,cv = skfold,verbose = 2)\n# grid_search.fit(train,y_train)\n# print(grid_search.best_params_)\n# print(grid_search.best_score_)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bafce1ea8895b941be959b011fecbddcb538f932"},"cell_type":"code","source":"import xgboost as xgb\nXGB = xgb.XGBClassifier(learning_rate = 0.05,n_jobs = -1,max_depth = 2,n_estimators = 200,\n                       subsample = 0.9,colsample_bytree = 0.9,min_child_weight = 1,\n                        gamma = 0.0,reg_alpha = 0.01,reg_lambda = 1,\n                        random_state = 66)   #parameters found by cross validation\nprint(\"XGB score : {:.4f}\".format(accuracy(XGB)))\nXGB.fit(train,y_train)\npred_xgb = XGB.predict(test)\nsubmission_xgb = pd.DataFrame({\"PassengerId\" : test_raw[\"PassengerId\"], \"Survived\" : pred_xgb })\nsubmission_xgb.to_csv(\"XGBoost\",index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"62cf948c233c183337360fdcb3fcf56b19b2bcb4"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}