{"cells":[{"metadata":{"trusted":false,"_uuid":"2bd415744f03d86333acbab6e183eb74d0da52b3"},"cell_type":"code","source":"# Imports\n\n# pandas\nimport pandas as pd\nfrom pandas import Series,DataFrame\n\n# numpy, matplotlib, seaborn\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"9df1ccb32f9616bf5a4355f732e78ae8aca4b26d"},"cell_type":"code","source":"# get data to dataframe variables\ntrain_df = pd.read_csv(\"../input/train.csv\")\ntest_df    = pd.read_csv(\"../input/test.csv\")\n\n# data visualization\ntrain_df\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4f7c619da45e10949525e638e940ce7f1f381ba3"},"cell_type":"code","source":"#inspecting train dataframe\ntrain_df.shape\ntrain_df.info()\ntrain_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"aa4f16e97aecb73a02b7022a91817a135a27ad05"},"cell_type":"code","source":"#inspecting test dataframe\ntest_df.shape\ntest_df.info()\ntest_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"0ca8228bb7cce1074a1a38b263ed2df30513a96d"},"cell_type":"code","source":"train_df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"fd7c84e952ab370349a2b1fcd91ce1a753fd790f"},"cell_type":"code","source":"# drop unnecessary columns: PassengerId, Name, Ticket, Cabin from train and test data\n#Note: Survived is not present in the test data\n\ntrain_df = train_df.drop(['PassengerId','Cabin','Ticket','Name'],axis=1)\n\ntest_PassengerId = test_df['PassengerId']\ntest_df = test_df.drop(['PassengerId','Cabin','Ticket','Name'],axis=1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"04bb0cc82a0f134f8120aba1eb528caa43d5e568"},"cell_type":"code","source":"print(test_df.columns)\ntrain_df.columns\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4721aa4d8aba910d1b099e0dfbac63ceed19a54e"},"cell_type":"code","source":"#checking the null value in the train and test data\nprint(train_df.isnull().sum())\nprint(test_df.isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"2cfab9bdb07ab7113d1c951728bc79bc383d3d3f"},"cell_type":"code","source":"# replacing the null age by median\ntrain_df['Age'].fillna(train_df['Age'].median(), inplace = True)\ntest_df['Age'].fillna(test_df['Age'].median(), inplace = True)\n\nprint(train_df.isnull().sum())\nprint(test_df.isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"7ea3acea1d038f61977780e4e66cc56260b03cb1"},"cell_type":"code","source":"train_df.Embarked.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"89ab7951f6de1901095cad9a83341580bbde2e92"},"cell_type":"code","source":"# Imputing the empty value with S in the embarked column\ntrain_df.loc[train_df.Embarked.isnull(),['Embarked']] ='S'\nprint(train_df.isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"b61fd0e2f9a04d80a1712a1b5241c7c954b68e51"},"cell_type":"code","source":"train_df.Embarked.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"7d480bb728493edc9c2b61db1866f9c84c3ae954"},"cell_type":"code","source":"# replacing the null fare by mean\ntest_df['Fare'].fillna(test_df['Fare'].mean(), inplace = True)\n\nprint(test_df.isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4f8347a20b8b448d41b28fbb64a2be5a10b40e41"},"cell_type":"code","source":"#Inspecting the datatype of the data\nprint(train_df.dtypes)\nprint(test_df.dtypes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"1433333f8448260892111108b28933719b26e280"},"cell_type":"code","source":"# Looking at the data distribution\ntrain_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4a5de91be182650e2140bed3435d4aef5d80a289"},"cell_type":"code","source":"# Looking at the data distribution\ntest_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"9c5f531b2688e7e8ca98a76b1be4d2bf3fa4ef0e"},"cell_type":"code","source":"# checking the distribution of the data\ntrain_df.hist()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"eb1478e2141417f6eb0b865f0938f25bf043f56c"},"cell_type":"code","source":"#Age\ntrain_df['Age'].hist()\ntrain_df['Age'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"2199d85af99f53c66f377337e134003be7967e67"},"cell_type":"code","source":"train_df.groupby('Sex')['Sex'].count()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"81a9c8c5605a198369aebac02ff2dbdbbe499db6"},"cell_type":"code","source":"sns.countplot('Embarked', data=train_df)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"7ff32ef7d08b1ae01efebc125e9d6797cc8dd23a"},"cell_type":"code","source":"train_df.groupby('Pclass')['Pclass'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"b47d1cbc3f913f2c4fadebcdbf1d6a2ec22a0f37"},"cell_type":"code","source":"def class_imput(x):\n    if(x==1):\n        return 'First'\n    elif(x==2):\n        return 'Second'\n    else:\n        return 'Third'\n    \ntrain_df.Pclass = train_df.Pclass.apply(class_imput)\ntest_df.Pclass = test_df.Pclass.apply(class_imput)\ntrain_df.head()                 ","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"aa6fd29c4eee63f61a60c6c3fc3ecdcf74670e68"},"cell_type":"code","source":"# Creating a dummy variable for some of the categorical variables and dropping the first one.\ndummy_train = pd.get_dummies(train_df[['Pclass', 'Sex', 'Embarked']], drop_first=True)\ndummy_test = pd.get_dummies(test_df[['Pclass', 'Sex', 'Embarked']], drop_first=True)\n# Adding the results to the master dataframe\ntrain_df = pd.concat([train_df, dummy_train], axis=1)\n    \ntest_df = pd.concat([test_df, dummy_test], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"d0480dcb7795022b5279eb785656227b3efec8a2"},"cell_type":"code","source":"# drop the original columns\ntrain_df = train_df.drop(['Pclass', 'Sex', 'Embarked'],axis=1)\ntest_df = test_df.drop(['Pclass', 'Sex', 'Embarked'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c3352a2199f8e68c78b0fb5fff67dba92743057b"},"cell_type":"code","source":"train_df.head()\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"dcf7589f88919e259bfb0fe3f5799434bc03e8fa"},"cell_type":"code","source":"train_df.groupby('SibSp')['SibSp'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"8835b2a4db71f491ea76e8d3727142eef0521dde"},"cell_type":"code","source":"train_df.groupby('Parch')['Parch'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"35d13cdab97d825869ccae0e0332a74194f10502"},"cell_type":"code","source":"# combining the columns SibSp and Parch and making one column family.\ntrain_df['Family'] = train_df['SibSp'] + train_df['Parch']\ntrain_df[train_df['Family'] > 0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"6610b3273521fd862bc2b5116079a42acef917ee"},"cell_type":"code","source":"train_df.groupby('Family')['Family'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"8c9c9d43264cb2ddf21b48e489cb1ccbcc4e9a5c"},"cell_type":"code","source":"# drop unnecessary columns SibSp and Parch now\ntrain_df = train_df.drop(['SibSp','Parch'], axis =1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"5f965f2024905debc987b5a0ad05e1e25c783cd7"},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4a5bde338fc8201acdd64c4a84ff52a6ccff6982"},"cell_type":"code","source":"# replacing non zero value by 1\ndef decode_family(x):\n    if x == 0:\n        return 0\n    else:\n        return 1\n    \n\n\ntrain_df['Family'] = train_df.Family.apply(decode_family)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ffe462478d277ab437c53b4fec969b94fc192831"},"cell_type":"code","source":"train_df.groupby('Family')['Family'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"beb0240010905da8c14515fe82f3e35a162635ba"},"cell_type":"code","source":"# Do same operation on test_df\n\n# combining the columns SibSp and Parch and making one column family.\ntest_df['Family'] = test_df['SibSp'] + test_df['Parch']\n\n# drop unnecessary columns SibSp and Parch now\ntest_df = test_df.drop(['SibSp','Parch'], axis =1)\n\ntest_df['Family'] = test_df.Family.apply(decode_family)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f072b3d608d6d0741ab2dd88d430378e340f1ba2"},"cell_type":"code","source":"test_df.info()\ntest_df.shape\ntest_df.groupby('Family')['Family'].count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"fcf4378b9cbaad89cdf55f3d56b88c2230d93f9e"},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4bdc6ae5f40829e99f31a2043dceb686a83aba81"},"cell_type":"code","source":"train_df.Survived.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"2abae5ebf9f6f01696d0822874d15f57a865b8db"},"cell_type":"code","source":"# separating dependent and independent variables\ny_train = train_df['Survived']\nX_train = train_df.drop('Survived',axis=1)\n\nX_test = test_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"92acb40420f7f8241a0b81cbcda7f933aeff5b66"},"cell_type":"code","source":"from sklearn import model_selection\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.svm import SVC\n\nfrom sklearn.ensemble import RandomForestClassifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"fee23cf28d4020896cd355447aec896aeb02d529"},"cell_type":"code","source":"# Test options and evaluation metric\nseed = 7\nscoring = 'accuracy'","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"a7c04f7be2fc8bea8906206ff15adb4a44b17f64"},"cell_type":"code","source":"# Spot Check Algorithms\nmodels = []\nmodels.append(('LR', LogisticRegression()))\nmodels.append(('LDA', LinearDiscriminantAnalysis()))\nmodels.append(('KNN', KNeighborsClassifier()))\nmodels.append(('CART', DecisionTreeClassifier()))\nmodels.append(('NB', GaussianNB()))\nmodels.append(('SVM', SVC()))\n# evaluate each model in turn\nresults = []\nnames = []\nfor name, model in models:\n    kfold = model_selection.KFold(n_splits=10, random_state=seed)\n    cv_results = model_selection.cross_val_score(model, X_train, y_train, cv=kfold, scoring=scoring)\n    results.append(cv_results)\n    names.append(name)\n    msg = \"%s: %f (%f)\" % (name, cv_results.mean(), cv_results.std())\n    print(msg)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"fe1f5c3270bea87031a601230bd7b9bb38118085"},"cell_type":"code","source":"fig = plt.figure()\nfig.suptitle('Algorithm Comparison')\nax = fig.add_subplot(111)\nplt.boxplot(results)\nax.set_xticklabels(names)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"48123fe877dc67b7e783e4df9b0cb1ca36f40671"},"cell_type":"code","source":"# Make predictions on validation dataset\nlda = LinearDiscriminantAnalysis()\nlda.fit(X_train, y_train)\npredictions = lda.predict(X_test)\n\npredictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c3695f33bfa6657e7c8defc25da2fef16ad2ddce"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"d5cbe18f667a37135ea3a72d25b6c51f544108b2"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.5"}},"nbformat":4,"nbformat_minor":1}