{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T00:52:41.694640Z","iopub.execute_input":"2022-07-12T00:52:41.695131Z","iopub.status.idle":"2022-07-12T00:52:41.705910Z","shell.execute_reply.started":"2022-07-12T00:52:41.695094Z","shell.execute_reply":"2022-07-12T00:52:41.704593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Importing libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:41.710655Z","iopub.execute_input":"2022-07-12T00:52:41.711049Z","iopub.status.idle":"2022-07-12T00:52:41.722473Z","shell.execute_reply.started":"2022-07-12T00:52:41.711006Z","shell.execute_reply":"2022-07-12T00:52:41.721175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:41.732905Z","iopub.execute_input":"2022-07-12T00:52:41.734184Z","iopub.status.idle":"2022-07-12T00:52:41.762638Z","shell.execute_reply.started":"2022-07-12T00:52:41.734130Z","shell.execute_reply":"2022-07-12T00:52:41.761245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data cleaning","metadata":{}},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:41.765026Z","iopub.execute_input":"2022-07-12T00:52:41.765411Z","iopub.status.idle":"2022-07-12T00:52:41.776705Z","shell.execute_reply.started":"2022-07-12T00:52:41.765376Z","shell.execute_reply":"2022-07-12T00:52:41.775335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:41.778644Z","iopub.execute_input":"2022-07-12T00:52:41.779367Z","iopub.status.idle":"2022-07-12T00:52:41.794732Z","shell.execute_reply.started":"2022-07-12T00:52:41.779333Z","shell.execute_reply":"2022-07-12T00:52:41.793466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing na values in Age with 999\ntrain_df[\"Age\"].fillna(999, inplace = True)\n# replacing na values in Cabin with Unknown\ntrain_df[\"Cabin\"].fillna(\"Unknown\", inplace = True)\n# replacing na values in Embarked with U\ntrain_df[\"Embarked\"].fillna(\"U\", inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:41.803831Z","iopub.execute_input":"2022-07-12T00:52:41.804294Z","iopub.status.idle":"2022-07-12T00:52:41.812811Z","shell.execute_reply.started":"2022-07-12T00:52:41.804250Z","shell.execute_reply":"2022-07-12T00:52:41.811909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:41.841538Z","iopub.execute_input":"2022-07-12T00:52:41.842253Z","iopub.status.idle":"2022-07-12T00:52:41.854075Z","shell.execute_reply.started":"2022-07-12T00:52:41.842211Z","shell.execute_reply":"2022-07-12T00:52:41.852567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ntrain_df.boxplot(column=['Fare', 'Age'], grid=False, color='black')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:41.884662Z","iopub.execute_input":"2022-07-12T00:52:41.885865Z","iopub.status.idle":"2022-07-12T00:52:42.083640Z","shell.execute_reply.started":"2022-07-12T00:52:41.885818Z","shell.execute_reply":"2022-07-12T00:52:42.082426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bar_plot = train_df['Pclass'].value_counts().plot(kind='bar')\nbar_plot.set_title('Distribution of passengers in each class')\nbar_plot.set_xlabel('Priority class')\nbar_plot.set_ylabel('Count')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:42.085614Z","iopub.execute_input":"2022-07-12T00:52:42.086051Z","iopub.status.idle":"2022-07-12T00:52:42.277399Z","shell.execute_reply.started":"2022-07-12T00:52:42.086018Z","shell.execute_reply":"2022-07-12T00:52:42.276026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.loc[train_df['Age'].between(0, 30, 'both'), 'Age category'] = 'Young'\ntrain_df.loc[train_df['Age'].between(31, 70, 'both'), 'Age category'] = 'Middle aged'\ntrain_df.loc[train_df['Age'].between(71, 100, 'both'), 'Age category'] = 'Old'\ntrain_df.loc[train_df['Age'].between(101, 1000, 'both'), 'Age category'] = 'Unknown'\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:42.278862Z","iopub.execute_input":"2022-07-12T00:52:42.279413Z","iopub.status.idle":"2022-07-12T00:52:42.309591Z","shell.execute_reply.started":"2022-07-12T00:52:42.279381Z","shell.execute_reply":"2022-07-12T00:52:42.308123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_bar_plot = train_df['Age category'].value_counts().plot(kind='bar')\nage_bar_plot.set_title('Distribution of Age')\nage_bar_plot.set_xlabel('Age categories')\nage_bar_plot.set_ylabel('Count')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:42.312456Z","iopub.execute_input":"2022-07-12T00:52:42.313098Z","iopub.status.idle":"2022-07-12T00:52:42.539614Z","shell.execute_reply.started":"2022-07-12T00:52:42.313060Z","shell.execute_reply":"2022-07-12T00:52:42.538570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns \nplt.figure(figsize=(10, 6))\nsns.scatterplot(x=\"Fare\", y=train_df.PassengerId, hue=\"Survived\", data=train_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:42.541194Z","iopub.execute_input":"2022-07-12T00:52:42.541586Z","iopub.status.idle":"2022-07-12T00:52:42.848357Z","shell.execute_reply.started":"2022-07-12T00:52:42.541550Z","shell.execute_reply":"2022-07-12T00:52:42.847106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"train_df_dropped_col = train_df.drop(['Name','Ticket','Cabin','Age'], axis=1)\ntrain_df_dropped_col.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:42.850260Z","iopub.execute_input":"2022-07-12T00:52:42.851498Z","iopub.status.idle":"2022-07-12T00:52:42.872128Z","shell.execute_reply.started":"2022-07-12T00:52:42.851444Z","shell.execute_reply":"2022-07-12T00:52:42.870888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# taking all rows but only 6 columns\ndf_small = train_df_dropped_col.iloc[:,:6]\ncorrelation_mat = df_small.corr()\nsns.heatmap(correlation_mat, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:42.874223Z","iopub.execute_input":"2022-07-12T00:52:42.875163Z","iopub.status.idle":"2022-07-12T00:52:43.206267Z","shell.execute_reply.started":"2022-07-12T00:52:42.875109Z","shell.execute_reply":"2022-07-12T00:52:43.204672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_dropped_with_dummies = pd.get_dummies(train_df_dropped_col)\ntrain_df_dropped_with_dummies.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.208438Z","iopub.execute_input":"2022-07-12T00:52:43.208930Z","iopub.status.idle":"2022-07-12T00:52:43.237099Z","shell.execute_reply.started":"2022-07-12T00:52:43.208893Z","shell.execute_reply":"2022-07-12T00:52:43.235885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.238759Z","iopub.execute_input":"2022-07-12T00:52:43.239192Z","iopub.status.idle":"2022-07-12T00:52:43.263651Z","shell.execute_reply.started":"2022-07-12T00:52:43.239157Z","shell.execute_reply":"2022-07-12T00:52:43.262420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_dropped_col = test_df.drop(['Name','Ticket','Cabin','Age'], axis=1)\ntest_df_dropped_with_dummies = pd.get_dummies(test_df_dropped_col)\ntest_df_dropped_with_dummies.loc[test_df['Age'].between(0, 30, 'both'), 'Age category'] = 'Young'\ntest_df_dropped_with_dummies.loc[test_df['Age'].between(31, 70, 'both'), 'Age category'] = 'Middle aged'\ntest_df_dropped_with_dummies.loc[test_df['Age'].between(71, 100, 'both'), 'Age category'] = 'Old'\ntest_df_dropped_with_dummies.loc[test_df['Age'].between(101, 1000, 'both'), 'Age category'] = 'Unknown'\ntest_df_dropped_with_dummies = pd.get_dummies(test_df_dropped_with_dummies)\ntest_df_dropped_with_dummies.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.267915Z","iopub.execute_input":"2022-07-12T00:52:43.268332Z","iopub.status.idle":"2022-07-12T00:52:43.311011Z","shell.execute_reply.started":"2022-07-12T00:52:43.268274Z","shell.execute_reply":"2022-07-12T00:52:43.309651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_dropped_with_dummies.head()\ntrain_df_dropped_with_dummies = train_df_dropped_with_dummies.drop(['Age category_Unknown','Embarked_U'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.315155Z","iopub.execute_input":"2022-07-12T00:52:43.315622Z","iopub.status.idle":"2022-07-12T00:52:43.324975Z","shell.execute_reply.started":"2022-07-12T00:52:43.315587Z","shell.execute_reply":"2022-07-12T00:52:43.323219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train_df_dropped_with_dummies.iloc[:,1:15]\ny = train_df_dropped_with_dummies['Survived']\n\nx=x.drop(['Survived'], axis=1)\n\nimport statsmodels.api as sm\nlogit_model=sm.Logit(y,x)\nresult=logit_model.fit()\nprint(result.summary2())","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.326945Z","iopub.execute_input":"2022-07-12T00:52:43.327615Z","iopub.status.idle":"2022-07-12T00:52:43.474117Z","shell.execute_reply.started":"2022-07-12T00:52:43.327563Z","shell.execute_reply":"2022-07-12T00:52:43.472718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_dropped_with_dummies.to_csv(\"dummydata.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.475992Z","iopub.execute_input":"2022-07-12T00:52:43.477252Z","iopub.status.idle":"2022-07-12T00:52:43.498183Z","shell.execute_reply.started":"2022-07-12T00:52:43.477194Z","shell.execute_reply":"2022-07-12T00:52:43.497217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_dropped_with_dummies.dtypes\ntrain_df_dropped_with_dummies['Sex_female'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.499860Z","iopub.execute_input":"2022-07-12T00:52:43.500497Z","iopub.status.idle":"2022-07-12T00:52:43.510875Z","shell.execute_reply.started":"2022-07-12T00:52:43.500459Z","shell.execute_reply":"2022-07-12T00:52:43.509661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic regression modelling","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn import metrics\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(x, y, test_size=0.3, random_state=0)\ny_test","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.512834Z","iopub.execute_input":"2022-07-12T00:52:43.513540Z","iopub.status.idle":"2022-07-12T00:52:43.525876Z","shell.execute_reply.started":"2022-07-12T00:52:43.513495Z","shell.execute_reply":"2022-07-12T00:52:43.524980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logreg = LogisticRegression()\nlogreg.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.527395Z","iopub.execute_input":"2022-07-12T00:52:43.527959Z","iopub.status.idle":"2022-07-12T00:52:43.571347Z","shell.execute_reply.started":"2022-07-12T00:52:43.527915Z","shell.execute_reply":"2022-07-12T00:52:43.570418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = logreg.predict(X_test)\nprint('Accuracy of logistic regression classifier on test set: {:.2f}'.format(logreg.score(X_test, y_test)))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.572752Z","iopub.execute_input":"2022-07-12T00:52:43.573364Z","iopub.status.idle":"2022-07-12T00:52:43.584155Z","shell.execute_reply.started":"2022-07-12T00:52:43.573328Z","shell.execute_reply":"2022-07-12T00:52:43.582859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Assumptions in the Logistic Regression Model\n1.Dependent Variable:Survival is binary  \n2.Very little multicolinerailty in the independent variables(Proved with the confusion matrix)","metadata":{}},{"cell_type":"markdown","source":"## Confusion Matrix","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nconfusion_matrix = confusion_matrix(y_test, y_pred)\nprint(confusion_matrix)\n#The result is telling us that we have 143+72 correct predictions and 25+28 incorrect predictions.","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.585911Z","iopub.execute_input":"2022-07-12T00:52:43.586284Z","iopub.status.idle":"2022-07-12T00:52:43.594754Z","shell.execute_reply.started":"2022-07-12T00:52:43.586249Z","shell.execute_reply":"2022-07-12T00:52:43.593656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.596005Z","iopub.execute_input":"2022-07-12T00:52:43.596941Z","iopub.status.idle":"2022-07-12T00:52:43.615803Z","shell.execute_reply.started":"2022-07-12T00:52:43.596901Z","shell.execute_reply":"2022-07-12T00:52:43.614343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ROC Curve","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import roc_curve\nlogit_roc_auc = roc_auc_score(y_test, logreg.predict(X_test))\nfpr, tpr, thresholds = roc_curve(y_test, logreg.predict_proba(X_test)[:,1])\nplt.figure()\nplt.plot(fpr, tpr, label='Logistic Regression (area = %0.2f)' % logit_roc_auc)\nplt.plot([0, 1], [0, 1],'r--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic')\nplt.legend(loc=\"lower right\")\nplt.savefig('Log_ROC')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.617147Z","iopub.execute_input":"2022-07-12T00:52:43.618620Z","iopub.status.idle":"2022-07-12T00:52:43.910888Z","shell.execute_reply.started":"2022-07-12T00:52:43.618559Z","shell.execute_reply":"2022-07-12T00:52:43.909985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df_dropped_with_dummies\nX_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.912294Z","iopub.execute_input":"2022-07-12T00:52:43.913643Z","iopub.status.idle":"2022-07-12T00:52:43.930694Z","shell.execute_reply.started":"2022-07-12T00:52:43.913590Z","shell.execute_reply":"2022-07-12T00:52:43.929621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exporting the final submission file for Logit","metadata":{}},{"cell_type":"code","source":"ids = X_test['PassengerId']\nX_test[\"Fare\"].fillna(0, inplace = True)\ny_test_pred = logreg.predict(X_test.drop('PassengerId', axis=1))\noutput = pd.DataFrame({ 'PassengerId' : ids, 'Survived': y_test_pred })\noutput.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.931873Z","iopub.execute_input":"2022-07-12T00:52:43.932668Z","iopub.status.idle":"2022-07-12T00:52:43.945752Z","shell.execute_reply.started":"2022-07-12T00:52:43.932632Z","shell.execute_reply":"2022-07-12T00:52:43.944392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## XG boost modelling","metadata":{}},{"cell_type":"code","source":"# fit model no training data\nfrom numpy import loadtxt\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nmodel = XGBClassifier()\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:43.947010Z","iopub.execute_input":"2022-07-12T00:52:43.947918Z","iopub.status.idle":"2022-07-12T00:52:44.308571Z","shell.execute_reply.started":"2022-07-12T00:52:43.947879Z","shell.execute_reply":"2022-07-12T00:52:44.307439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df_dropped_with_dummies\nmodel = XGBClassifier()\nmodel.fit(X_train, y_train)\nids = X_test['PassengerId']\n# make predictions for test data\ny_pred = model.predict(X_test.drop('PassengerId', axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:44.312994Z","iopub.execute_input":"2022-07-12T00:52:44.314066Z","iopub.status.idle":"2022-07-12T00:52:44.641533Z","shell.execute_reply.started":"2022-07-12T00:52:44.314021Z","shell.execute_reply":"2022-07-12T00:52:44.640488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:44.646477Z","iopub.execute_input":"2022-07-12T00:52:44.647240Z","iopub.status.idle":"2022-07-12T00:52:44.659759Z","shell.execute_reply.started":"2022-07-12T00:52:44.647197Z","shell.execute_reply":"2022-07-12T00:52:44.658289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\npredictions = [round(value) for value in y_pred]\noutput = pd.DataFrame({ 'PassengerId' : ids, 'Survived': predictions })\noutput.to_csv('submissionXGBoost.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:44.661706Z","iopub.execute_input":"2022-07-12T00:52:44.663882Z","iopub.status.idle":"2022-07-12T00:52:44.676747Z","shell.execute_reply.started":"2022-07-12T00:52:44.663807Z","shell.execute_reply":"2022-07-12T00:52:44.674830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random forest modelling","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nmodel = RandomForestClassifier(n_estimators=100, max_depth=5, random_state=1)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:44.679467Z","iopub.execute_input":"2022-07-12T00:52:44.680108Z","iopub.status.idle":"2022-07-12T00:52:44.929553Z","shell.execute_reply.started":"2022-07-12T00:52:44.680061Z","shell.execute_reply":"2022-07-12T00:52:44.928153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test[\"Fare\"].fillna(0, inplace = True)\npredictions = model.predict(X_test.drop('PassengerId', axis=1))\npredictions","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:44.934670Z","iopub.execute_input":"2022-07-12T00:52:44.935090Z","iopub.status.idle":"2022-07-12T00:52:44.970298Z","shell.execute_reply.started":"2022-07-12T00:52:44.935055Z","shell.execute_reply":"2022-07-12T00:52:44.968909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Accuracy of random forest","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:44.972975Z","iopub.execute_input":"2022-07-12T00:52:44.973537Z","iopub.status.idle":"2022-07-12T00:52:44.980240Z","shell.execute_reply.started":"2022-07-12T00:52:44.973485Z","shell.execute_reply":"2022-07-12T00:52:44.978526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({ 'PassengerId' : ids, 'Survived': predictions })\noutput.to_csv('submissionRandomForest.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:52:44.982303Z","iopub.execute_input":"2022-07-12T00:52:44.983180Z","iopub.status.idle":"2022-07-12T00:52:45.000473Z","shell.execute_reply.started":"2022-07-12T00:52:44.983128Z","shell.execute_reply":"2022-07-12T00:52:44.999084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## KNN Modelling","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nclassifier = KNeighborsClassifier(n_neighbors=5)\nclassifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:53:44.265930Z","iopub.execute_input":"2022-07-12T01:53:44.266736Z","iopub.status.idle":"2022-07-12T01:53:44.297322Z","shell.execute_reply.started":"2022-07-12T01:53:44.266690Z","shell.execute_reply":"2022-07-12T01:53:44.295952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = classifier.predict(X_test.drop('PassengerId', axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:54:33.394496Z","iopub.execute_input":"2022-07-12T01:54:33.394916Z","iopub.status.idle":"2022-07-12T01:54:33.430019Z","shell.execute_reply.started":"2022-07-12T01:54:33.394885Z","shell.execute_reply":"2022-07-12T01:54:33.428967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({ 'PassengerId' : ids, 'Survived': predictions })\noutput.to_csv('submissionKNN.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:55:52.589494Z","iopub.execute_input":"2022-07-12T01:55:52.590406Z","iopub.status.idle":"2022-07-12T01:55:52.603734Z","shell.execute_reply.started":"2022-07-12T01:55:52.590352Z","shell.execute_reply":"2022-07-12T01:55:52.602404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SVM\n","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\n\n# Training a SVM classifier using SVC class\nsvm = SVC(kernel= 'linear', random_state=1, C=0.1)\nsvm.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:02:48.566441Z","iopub.execute_input":"2022-07-12T02:02:48.567332Z","iopub.status.idle":"2022-07-12T02:02:48.693326Z","shell.execute_reply.started":"2022-07-12T02:02:48.567268Z","shell.execute_reply":"2022-07-12T02:02:48.692426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX_test = test_df_dropped_with_dummies\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:03:35.952020Z","iopub.execute_input":"2022-07-12T02:03:35.952497Z","iopub.status.idle":"2022-07-12T02:03:35.969604Z","shell.execute_reply.started":"2022-07-12T02:03:35.952460Z","shell.execute_reply":"2022-07-12T02:03:35.967879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = svm.predict(X_test.drop('PassengerId', axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:04:20.790811Z","iopub.execute_input":"2022-07-12T02:04:20.791247Z","iopub.status.idle":"2022-07-12T02:04:20.804725Z","shell.execute_reply.started":"2022-07-12T02:04:20.791212Z","shell.execute_reply":"2022-07-12T02:04:20.803679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({ 'PassengerId' : ids, 'Survived': predictions })\noutput.to_csv('submissionSVM.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:04:48.806553Z","iopub.execute_input":"2022-07-12T02:04:48.806982Z","iopub.status.idle":"2022-07-12T02:04:48.816098Z","shell.execute_reply.started":"2022-07-12T02:04:48.806948Z","shell.execute_reply":"2022-07-12T02:04:48.815090Z"},"trusted":true},"execution_count":null,"outputs":[]}]}