{"cells":[{"metadata":{"_uuid":"66f8df48f7d5e8e2f6d3c44d7caea3c7644623e0"},"cell_type":"markdown","source":"# Case Study For The Titanic Competition (Data Analyse&Logistic Regression)\n\n# Step 1 Importing Necessary Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"#import os\n#print(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ed9bc286471aabba8b2e32f6befc9755c011d676"},"cell_type":"markdown","source":"# Step 2 Loading Data"},{"metadata":{"trusted":true,"_uuid":"c636bdcb5947666cd10a168f9f453c6f558bffa1"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1dc7c2cfc4fb2d00a950dbe01d1358e7a9259d14"},"cell_type":"code","source":"train.head(4)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"340f345e60d2441eb1add4b00fa4bebcbc1f5070"},"cell_type":"markdown","source":"# Step 3 Exploratory Data Analysis"},{"metadata":{"trusted":true,"_uuid":"1f09261ff3d8ed70595d3b26e3cb081e099d286a"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"95c94753ea6cf79b6ef5b25db0e5829c256c2c18"},"cell_type":"code","source":"train.isnull().sample(25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f21adfd18ad4fab568d1871f00f63aecc5167138"},"cell_type":"code","source":"sns.heatmap(train.isnull(), yticklabels=False, cbar=False, cmap=\"viridis\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"135e3732b1e3359a2979df362417b12a7bcde730"},"cell_type":"markdown","source":"As seen above Age column and Cabin column have lots of missing information (NaN values)."},{"metadata":{"trusted":true,"_uuid":"0b50bad64fd7aa0ff52a3c4fbfed4948706f1839"},"cell_type":"code","source":"sns.set_style('whitegrid')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b72dfdfc12b864fcbe36827fee551a3d80632d6a"},"cell_type":"code","source":"sns.countplot(x='Survived',hue='Sex', data=train, palette='RdBu_r')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"165e0dac36ac19a5155cba6ce9f580b5b41c6853"},"cell_type":"markdown","source":"It looks like people that did not survive(0) were much more likely to be male and people that survive were mostly female."},{"metadata":{"trusted":true,"_uuid":"e3ef9f8ae38af34a8def57c13943633b7ac0cb3f"},"cell_type":"code","source":"sns.countplot(x='Survived',hue='Pclass', data=train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bcf0660ae08f106a372dfe01cdf6fe624d84f8e6"},"cell_type":"markdown","source":"We can see above chart people who did not survive overwhelmingly part of the third class."},{"metadata":{"trusted":true,"_uuid":"ffe8b617a97368b3447bf211c649c355b9e7d7a3"},"cell_type":"code","source":"sns.distplot(train['Age'].dropna(), kde=False, bins=30 )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ed4b8f8d7cc000caae4e06fea6a7751438aad5ed"},"cell_type":"markdown","source":"It is quite skewed towards younger passengers."},{"metadata":{"trusted":true,"_uuid":"db50bebcfdd9b332e6ce8ccd0a9f5b0ca2ea4164"},"cell_type":"code","source":"sns.countplot(x='SibSp', data=train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"74633b7e9beeee0e4fd8ebd94ace3bcd296421ce"},"cell_type":"markdown","source":"The accounts of siblings versus spouses on board. (Most of them were probably single)"},{"metadata":{"trusted":true,"_uuid":"daa3665108d476c2f3b9685e5b7ed404276a37af"},"cell_type":"code","source":"train['Fare'].hist(bins=50, figsize=(10,4))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b5c0f7b90430e74ce454869b087a89ef258f985d"},"cell_type":"markdown","source":"It looks like most of the prices are between 0 and 50$ at that time (1912) ."},{"metadata":{"_uuid":"28a08a201bb3aab3de04c91879ef74a98d66ef3c"},"cell_type":"markdown","source":"# Step 4 Cleanin Data"},{"metadata":{"_uuid":"315ec3c5f4699b70d6d3cef5536402aabe0f98d6"},"cell_type":"markdown","source":"Instead of dropping all the missing values we can do some better touches. To be specifik, impulation or averege age can be used to fill in missing data for the 'Age' column. "},{"metadata":{"trusted":true,"_uuid":"27b293f12ef833660ce9e3d2dda436f4fad5d18e"},"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.boxplot(x='Pclass', y='Age', data=train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"517dca0717603cc50260c59a0d5c0cd95159dd07"},"cell_type":"markdown","source":"It seems the more you older the more wealthy. We can use these average age values in order to impute the age based off the passenger class. "},{"metadata":{"trusted":true,"_uuid":"0b59be5b1e2d0763a4821f518525d8ca7c347b42"},"cell_type":"code","source":"def impute_average(columns):\n    Age = columns[0]\n    Pclass = columns[1]\n    if pd.isnull(Age):\n        if Pclass == 1:\n            return 37\n        elif Pclass == 2:\n            return 29\n        else:\n            return 24\n    else:\n        return Age","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a88961f393b04e1f3f5113546368ac542003b08d"},"cell_type":"code","source":"train['Age'] = train[['Age','Pclass']].apply(impute_average, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"08c04f8ada99c57d09da683827ece3e2bd8092e0"},"cell_type":"code","source":"sns.heatmap(train.isnull(), yticklabels=False, cbar=False, cmap='viridis')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bc7ba59b9fd50166e5339744b62da03ed5d88640"},"cell_type":"markdown","source":"It looks like we are no longer having any missing information for the age column.\nOn the other hand there is so much missing points for the Cabin column. Best action for this column is drop that cabin."},{"metadata":{"trusted":true,"_uuid":"06cb22893dfd01125abd0f488495faf12aabd8be"},"cell_type":"code","source":"train.drop('Cabin', axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"307b2d3cffb538cbd0990bbd4c2b686a75d1cc29"},"cell_type":"code","source":"train.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"27ea9ff745d66ad13898eaa099e04e43051d2ae3"},"cell_type":"code","source":"#there are few missing values to get rid of them we drop them.\ntrain.dropna(inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a3ee2572bc84b67a5d616ec8206dffb08440f47"},"cell_type":"code","source":"sns.heatmap(train.isnull(), yticklabels=False, cbar=False, cmap='viridis')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8d7c7fbe1e59bfbb4a2a5ea653db7d72a3575550"},"cell_type":"markdown","source":"# Step 5 Feature Engineering"},{"metadata":{"_uuid":"2154d62944b67f09023a903ffb37e1a96b274667"},"cell_type":"markdown","source":"We need to preapare data to ML algorithm in order to that we should deal some of the data features. \nIn this context we need to convert some columns, categorical features, into dummy variables, and drop some of the columns which we won't use."},{"metadata":{"trusted":true,"_uuid":"dcaa58930a7470585e0f09b670961259ece69aaa"},"cell_type":"code","source":"sex = pd.get_dummies(train['Sex'], drop_first=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20b643fb122e760bf6e2fea0a0d0b4b3e2af3483"},"cell_type":"code","source":"embark = pd.get_dummies(train['Embarked'], drop_first=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9886140e4bbe6ba68d6fc25e5f1d2443a6835379"},"cell_type":"code","source":"train = pd.concat([train,sex,embark],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"618d51116e1f5e57955186f8ef234338d9d21b3f"},"cell_type":"code","source":"train.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8fe64838041413f596cc61e6b8d214a47a0b0fe0"},"cell_type":"code","source":"train.drop(['Sex', 'Embarked', 'Name','Ticket'],axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16394d8defc225c9dcc86b972f5652ca3135a5fe"},"cell_type":"code","source":"train.head(3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"959c06e139b02776aad97d16f6cc751e85e085eb"},"cell_type":"markdown","source":"One last thing for this step is PassengerId column. It is smilar to index column and we won't use in our model for that reason we can drop it. "},{"metadata":{"trusted":true,"_uuid":"aae5a90deb154160aacbc5d51f7693f6e5b67d3c"},"cell_type":"code","source":"train.drop('PassengerId', axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"929c7a51122583307315a6a3d615ac97026f98b5"},"cell_type":"code","source":"train.head(3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"567173c78d9ee468e488748ddb5ed9b62139d563"},"cell_type":"markdown","source":"# Step 6 Training the Model"},{"metadata":{"_uuid":"2521784ba9281341c500636b27a30ee77631bea0"},"cell_type":"markdown","source":"I choose to use Logistic Regression. You can try any convenient ML model."},{"metadata":{"_uuid":"f26e1b4c8ef05dad7fcb29309b29da2791e68385"},"cell_type":"markdown","source":"## a. Selecting Train Target Data Set and Spliting Them."},{"metadata":{"trusted":true,"_uuid":"f1c63bab99e9eab1d66b15f0629d389990277dcb"},"cell_type":"code","source":"X = train.drop('Survived', axis=1)\ny = train['Survived']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9495db94e88ef16fde0ba5f483b60a975073c7d"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d096509ce2580f446c8a6576ea7e6278f46fac3b"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=101)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a07e51d75a07a102049a2526755d031b1281ee0d"},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5852288b70686800396818f0ae77fdbf387250a6"},"cell_type":"code","source":"model = LogisticRegression()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46d22fc3513ee4c5b5dc42fbd4ccc709645a2277"},"cell_type":"code","source":"model.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"69a8b9c0ae083878518c9be404600205a3ce52c1"},"cell_type":"markdown","source":"# Step 7 Predicting the Model"},{"metadata":{"trusted":true,"_uuid":"0e81fd66b427648d4f2fedbbd142175a06dcf6ca"},"cell_type":"code","source":"predictions = model.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60c1d8229e8360643eb86694e1795d9981540129"},"cell_type":"code","source":"from sklearn.metrics import classification_report","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ad57a04c81333ee3fe73e747ca7da916ff63730","scrolled":true},"cell_type":"code","source":"print(classification_report(y_test, predictions))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"3915f8b8bb0b7c37058aa4d63733a1a819568e74"},"cell_type":"code","source":"from sklearn.metrics import confusion_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b145bc1fe673c931b85b6e660c7a63915d7594a3"},"cell_type":"code","source":"confusion_matrix(y_test, predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eb57504f847b5b3dbde3668f5bbf38aeb5eccfb6"},"cell_type":"code","source":"sns.heatmap(confusion_matrix(y_test, predictions),annot=True,fmt=\"d\",cmap='ocean_r' ,robust=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d73d22a983880d706fe7a89ac70888de7d59142e"},"cell_type":"code","source":"# Lastly we can strengthen the results with cross_val_score\nfrom sklearn.model_selection import cross_val_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d3ba7d84d5a073945e126caacded19554370a3a5"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e18752813c4033ad6385e621ff49f21f0e5271ca"},"cell_type":"code","source":"print(cross_val_score(model, X, y, cv=19).mean())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0f5b31c8ec22937eabe993020393563405c4080a"},"cell_type":"markdown","source":"# Great Job!"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}