{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Data Analysis and Wrangling\nimport pandas as pd\nimport numpy as np\nimport random as rnd\nimport os\n\n#Visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n#Machine Learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron, SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:17.787771Z","iopub.execute_input":"2022-07-28T07:34:17.788685Z","iopub.status.idle":"2022-07-28T07:34:17.797647Z","shell.execute_reply.started":"2022-07-28T07:34:17.788641Z","shell.execute_reply":"2022-07-28T07:34:17.796437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cwd = os.getcwd()\ncwd = '/kaggle/input'\nTRAIN_PATH = '/titanic/train.csv'\nTEST_PATH = '/titanic/test.csv'\nGENDER_SUBMISSION_PATH = '/titanic/gender_submission.csv'\nSUBMISSION_PATH = '/titanic/submission.csv'\ntrain_df = pd.read_csv(cwd+TRAIN_PATH)\ntest_df = pd.read_csv(cwd+TEST_PATH)\ncombine = [ train_df, test_df]","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:17.820099Z","iopub.execute_input":"2022-07-28T07:34:17.820619Z","iopub.status.idle":"2022-07-28T07:34:17.835813Z","shell.execute_reply.started":"2022-07-28T07:34:17.820591Z","shell.execute_reply":"2022-07-28T07:34:17.835185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.info())\nprint(\"_\"*40)\nprint(test_df.info())\n#Categorical - Survived, Sex, Embarked\n#Ordinal - PClass\n#Continous - Age, Fare\n#Discrete - SibSp, Parch","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:17.847239Z","iopub.execute_input":"2022-07-28T07:34:17.848071Z","iopub.status.idle":"2022-07-28T07:34:17.868689Z","shell.execute_reply.started":"2022-07-28T07:34:17.848030Z","shell.execute_reply":"2022-07-28T07:34:17.867881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:17.884582Z","iopub.execute_input":"2022-07-28T07:34:17.885019Z","iopub.status.idle":"2022-07-28T07:34:17.914139Z","shell.execute_reply.started":"2022-07-28T07:34:17.884989Z","shell.execute_reply":"2022-07-28T07:34:17.913238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.corr()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:17.917743Z","iopub.execute_input":"2022-07-28T07:34:17.918185Z","iopub.status.idle":"2022-07-28T07:34:17.931002Z","shell.execute_reply.started":"2022-07-28T07:34:17.918159Z","shell.execute_reply":"2022-07-28T07:34:17.930175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#High Correlation among PClass=1 and Survived\ntrain_df[['Pclass', 'Survived']].groupby(['Pclass'], as_index=False).mean().sort_values(by='Survived', ascending = False)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:17.958405Z","iopub.execute_input":"2022-07-28T07:34:17.958973Z","iopub.status.idle":"2022-07-28T07:34:17.972047Z","shell.execute_reply.started":"2022-07-28T07:34:17.958940Z","shell.execute_reply":"2022-07-28T07:34:17.970865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Female had very high survival rate\ntrain_df[['Sex', 'Survived']].groupby(['Sex'], as_index=False).mean().sort_values(by='Survived', ascending = False)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:18.009152Z","iopub.execute_input":"2022-07-28T07:34:18.009659Z","iopub.status.idle":"2022-07-28T07:34:18.022640Z","shell.execute_reply.started":"2022-07-28T07:34:18.009631Z","shell.execute_reply":"2022-07-28T07:34:18.021556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = sns.FacetGrid(train_df, col = 'Survived')\ng.map(plt.hist, 'Age', bins=20)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:18.035334Z","iopub.execute_input":"2022-07-28T07:34:18.035610Z","iopub.status.idle":"2022-07-28T07:34:18.410433Z","shell.execute_reply.started":"2022-07-28T07:34:18.035587Z","shell.execute_reply":"2022-07-28T07:34:18.409453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pclass was intrumental in Surviving Titanic\ngrid = sns.FacetGrid(train_df, col='Survived', row='Pclass', size=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:18.412069Z","iopub.execute_input":"2022-07-28T07:34:18.412359Z","iopub.status.idle":"2022-07-28T07:34:19.563833Z","shell.execute_reply.started":"2022-07-28T07:34:18.412334Z","shell.execute_reply":"2022-07-28T07:34:19.562839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Sex feature import for model\ngrid = sns.FacetGrid(train_df, row='Embarked', size=2.2, aspect=1.6)\ngrid.map(sns.pointplot, 'Pclass', 'Survived', 'Sex', palette='deep')\ngrid.add_legend()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:19.565197Z","iopub.execute_input":"2022-07-28T07:34:19.565486Z","iopub.status.idle":"2022-07-28T07:34:20.562867Z","shell.execute_reply.started":"2022-07-28T07:34:19.565460Z","shell.execute_reply":"2022-07-28T07:34:20.561855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fare Important Feature\ngrid = sns.FacetGrid(train_df, row='Embarked', col='Survived', height=2.2, aspect=1.6)\ngrid.map(sns.barplot, 'Sex', 'Fare', ci=None)\ngrid.add_legend()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:20.565555Z","iopub.execute_input":"2022-07-28T07:34:20.566111Z","iopub.status.idle":"2022-07-28T07:34:21.486836Z","shell.execute_reply.started":"2022-07-28T07:34:20.566069Z","shell.execute_reply":"2022-07-28T07:34:21.485865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ticket and Cabin have no relationship with Survival\ntrain_df = train_df.drop(['Ticket', 'Cabin'], axis=1)\ntest_df = test_df.drop(['Ticket', 'Cabin'], axis=1)\ncombine = [train_df, test_df]","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:21.488092Z","iopub.execute_input":"2022-07-28T07:34:21.488379Z","iopub.status.idle":"2022-07-28T07:34:21.495103Z","shell.execute_reply.started":"2022-07-28T07:34:21.488352Z","shell.execute_reply":"2022-07-28T07:34:21.494143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train_df['Title'], train_df['Sex'])","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:21.496183Z","iopub.execute_input":"2022-07-28T07:34:21.496442Z","iopub.status.idle":"2022-07-28T07:34:21.523886Z","shell.execute_reply.started":"2022-07-28T07:34:21.496417Z","shell.execute_reply":"2022-07-28T07:34:21.522962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Title is useful. Let's keep it and clean it up\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Countess', 'Capt', 'Col', 'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n\ntitle_mapping = {\"Mr\":1, \"Miss\":2, \"Mrs\":3, \"Master\":4, \"Rare\":5}\n\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].map(title_mapping)\n    dataset['Title'] = dataset['Title'].fillna(0)\n\ntrain_df[['Title', 'Survived']].groupby(['Title'], as_index=False).mean()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:21.525127Z","iopub.execute_input":"2022-07-28T07:34:21.525386Z","iopub.status.idle":"2022-07-28T07:34:21.548060Z","shell.execute_reply.started":"2022-07-28T07:34:21.525362Z","shell.execute_reply":"2022-07-28T07:34:21.547171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Dropping Name and PassengerId as we have Title now\ntrain_df = train_df.drop(['Name', 'PassengerId'], axis=1)\ntest_df = test_df.drop(['Name'], axis=1)\ncombine = [train_df, test_df]","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:21.549292Z","iopub.execute_input":"2022-07-28T07:34:21.549553Z","iopub.status.idle":"2022-07-28T07:34:21.555798Z","shell.execute_reply.started":"2022-07-28T07:34:21.549528Z","shell.execute_reply":"2022-07-28T07:34:21.555140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting Categorical data to int\nfor dataset in combine:\n    dataset['Sex'] = dataset['Sex'].map({\n        'female' : 1,\n        'male' : 0\n    }).astype(int)\ntrain_df.head()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:21.557030Z","iopub.execute_input":"2022-07-28T07:34:21.557384Z","iopub.status.idle":"2022-07-28T07:34:21.579299Z","shell.execute_reply.started":"2022-07-28T07:34:21.557346Z","shell.execute_reply":"2022-07-28T07:34:21.578254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid = sns.FacetGrid(train_df, row='Pclass', col='Sex', size=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:21.582525Z","iopub.execute_input":"2022-07-28T07:34:21.582930Z","iopub.status.idle":"2022-07-28T07:34:22.730932Z","shell.execute_reply.started":"2022-07-28T07:34:21.582892Z","shell.execute_reply":"2022-07-28T07:34:22.730051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fix Age which is NULL\nguess_ages = np.zeros((2,3))\nfor dataset in combine:\n    for i in range(0,2):\n        for j in range(0,3):\n            guess_df = dataset[\n                (dataset['Sex'] == i) &\n                (dataset['Pclass'] == j+1)\n            ]['Age'].dropna()\n            age_guess = guess_df.median()\n            guess_ages[i,j] = int (age_guess/0.5+0.5)*0.5\n\n    for i in range(0,2):\n        for j in range(0,3):\n            dataset.loc[ (dataset.Age.isnull()) & (dataset.Sex ==i) & (dataset.Pclass == j+1), 'Age'] = guess_ages[i,j]\n\n    dataset['Age'] = dataset['Age'].astype(int)\n\ntrain_df.head()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.732517Z","iopub.execute_input":"2022-07-28T07:34:22.733166Z","iopub.status.idle":"2022-07-28T07:34:22.774183Z","shell.execute_reply.started":"2022-07-28T07:34:22.733126Z","shell.execute_reply":"2022-07-28T07:34:22.773024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['AgeBand'] = pd.cut(train_df['Age'], 5)\ntrain_df[['AgeBand', 'Survived']].groupby(['AgeBand'], as_index=False).mean().sort_values(by='AgeBand', ascending=True)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.777206Z","iopub.execute_input":"2022-07-28T07:34:22.777579Z","iopub.status.idle":"2022-07-28T07:34:22.794930Z","shell.execute_reply.started":"2022-07-28T07:34:22.777549Z","shell.execute_reply":"2022-07-28T07:34:22.793959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset.loc[ dataset['Age'] <= 16, 'Age'] = 0\n    dataset.loc[(dataset['Age'] > 16) & (dataset['Age'] <= 32), 'Age'] = 1\n    dataset.loc[(dataset['Age'] > 32) & (dataset['Age'] <= 48), 'Age'] = 2\n    dataset.loc[(dataset['Age'] > 48) & (dataset['Age'] <= 64), 'Age'] = 3\n    dataset.loc[ dataset['Age'] > 64, 'Age']\ntrain_df.head()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.796147Z","iopub.execute_input":"2022-07-28T07:34:22.796430Z","iopub.status.idle":"2022-07-28T07:34:22.819340Z","shell.execute_reply.started":"2022-07-28T07:34:22.796405Z","shell.execute_reply":"2022-07-28T07:34:22.818497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(['AgeBand'], axis=1)\ncombine = [train_df, test_df]","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.820343Z","iopub.execute_input":"2022-07-28T07:34:22.820621Z","iopub.status.idle":"2022-07-28T07:34:22.825653Z","shell.execute_reply.started":"2022-07-28T07:34:22.820596Z","shell.execute_reply":"2022-07-28T07:34:22.824787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['FamilySize'] = dataset['SibSp'] + dataset['Parch'] + 1\n\ntrain_df[['FamilySize', 'Survived']].groupby(['FamilySize'], as_index=False).mean().sort_values(by = 'Survived', ascending=False)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.826955Z","iopub.execute_input":"2022-07-28T07:34:22.827312Z","iopub.status.idle":"2022-07-28T07:34:22.845207Z","shell.execute_reply.started":"2022-07-28T07:34:22.827288Z","shell.execute_reply":"2022-07-28T07:34:22.844332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['IsAlone'] = 0\n    dataset.loc[dataset['FamilySize'] == 1, 'IsAlone'] = 1\n\ntrain_df[['IsAlone', 'Survived']].groupby(['IsAlone'], as_index=False).mean()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.846220Z","iopub.execute_input":"2022-07-28T07:34:22.846491Z","iopub.status.idle":"2022-07-28T07:34:22.861642Z","shell.execute_reply.started":"2022-07-28T07:34:22.846466Z","shell.execute_reply":"2022-07-28T07:34:22.860721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(['Parch', 'SibSp', 'FamilySize'], axis=1)\ntest_df = test_df.drop(['Parch', 'SibSp', 'FamilySize'], axis=1)\ncombine = [train_df, test_df]","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.863165Z","iopub.execute_input":"2022-07-28T07:34:22.863533Z","iopub.status.idle":"2022-07-28T07:34:22.870150Z","shell.execute_reply.started":"2022-07-28T07:34:22.863496Z","shell.execute_reply":"2022-07-28T07:34:22.869312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create Artifical Feature combining PClass and Age\n\nfor dataset in combine:\n    dataset['Age*Class'] = dataset.Age * dataset.Pclass","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.871264Z","iopub.execute_input":"2022-07-28T07:34:22.871530Z","iopub.status.idle":"2022-07-28T07:34:22.882452Z","shell.execute_reply.started":"2022-07-28T07:34:22.871505Z","shell.execute_reply":"2022-07-28T07:34:22.881837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freq_port = train_df.Embarked.dropna().mode()[0]","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.883534Z","iopub.execute_input":"2022-07-28T07:34:22.884006Z","iopub.status.idle":"2022-07-28T07:34:22.893603Z","shell.execute_reply.started":"2022-07-28T07:34:22.883979Z","shell.execute_reply":"2022-07-28T07:34:22.892773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].fillna(freq_port)\n    dataset['Embarked'] = dataset['Embarked'].map(\n        {'S':0, 'C':1, 'Q':2}\n    ).astype(int)\n\ntrain_df[['Embarked', 'Survived']].groupby(['Embarked'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.894597Z","iopub.execute_input":"2022-07-28T07:34:22.894881Z","iopub.status.idle":"2022-07-28T07:34:22.913359Z","shell.execute_reply.started":"2022-07-28T07:34:22.894854Z","shell.execute_reply":"2022-07-28T07:34:22.912617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['Fare'].fillna(test_df['Fare'].dropna().median(), inplace=True)\ntest_df.head()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.914273Z","iopub.execute_input":"2022-07-28T07:34:22.915096Z","iopub.status.idle":"2022-07-28T07:34:22.926578Z","shell.execute_reply.started":"2022-07-28T07:34:22.915069Z","shell.execute_reply":"2022-07-28T07:34:22.925957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['FareBand'] = pd.qcut(train_df['Fare'], 4)\ntrain_df[['FareBand', 'Survived']].groupby(['FareBand'], as_index=False).mean().sort_values(by='FareBand', ascending=True)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.927480Z","iopub.execute_input":"2022-07-28T07:34:22.928095Z","iopub.status.idle":"2022-07-28T07:34:22.947717Z","shell.execute_reply.started":"2022-07-28T07:34:22.928065Z","shell.execute_reply":"2022-07-28T07:34:22.946809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset.loc[ dataset['Fare'] <= 7.91, 'Fare'] = 0\n    dataset.loc[(dataset['Fare'] > 7.91) & (dataset['Fare'] <= 14.454), 'Fare'] = 1\n    dataset.loc[(dataset['Fare'] > 14.454) & (dataset['Fare'] <= 31), 'Fare']   = 2\n    dataset.loc[ dataset['Fare'] > 31, 'Fare'] = 3\n    dataset['Fare'] = dataset['Fare'].astype(int)\n\ntrain_df = train_df.drop(['FareBand'], axis=1)\ncombine = [train_df, test_df]\n\ntrain_df.head(10)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.948873Z","iopub.execute_input":"2022-07-28T07:34:22.949137Z","iopub.status.idle":"2022-07-28T07:34:22.969255Z","shell.execute_reply.started":"2022-07-28T07:34:22.949113Z","shell.execute_reply":"2022-07-28T07:34:22.968353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Woohooo Let's solve this bad boy\n\nX_train = train_df.drop(\"Survived\", axis=1)\nY_train = train_df[\"Survived\"]\nX_test = test_df.drop(\"PassengerId\", axis=1).copy()\nX_train.shape, Y_train.shape, X_test.shape","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.970485Z","iopub.execute_input":"2022-07-28T07:34:22.970763Z","iopub.status.idle":"2022-07-28T07:34:22.978388Z","shell.execute_reply.started":"2022-07-28T07:34:22.970731Z","shell.execute_reply":"2022-07-28T07:34:22.977774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.info()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.979271Z","iopub.execute_input":"2022-07-28T07:34:22.980072Z","iopub.status.idle":"2022-07-28T07:34:22.993123Z","shell.execute_reply.started":"2022-07-28T07:34:22.980046Z","shell.execute_reply":"2022-07-28T07:34:22.992059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"pycharm":{"name":"#%% md\n"}}},{"cell_type":"code","source":"logreg = LogisticRegression()\nlogreg.fit(X_train, Y_train)\nY_pred = logreg.predict(X_test)\nacc_log = round(logreg.score(X_train, Y_train)*100,2)\nacc_log","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:22.994457Z","iopub.execute_input":"2022-07-28T07:34:22.994713Z","iopub.status.idle":"2022-07-28T07:34:23.018907Z","shell.execute_reply.started":"2022-07-28T07:34:22.994688Z","shell.execute_reply":"2022-07-28T07:34:23.017988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coeff_df = pd.DataFrame(train_df.columns.delete(0))\ncoeff_df.columns = ['Feature']\ncoeff_df['Correlation'] = pd.Series(logreg.coef_[0])\ncoeff_df.sort_values(by = 'Correlation', ascending=False)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.020185Z","iopub.execute_input":"2022-07-28T07:34:23.020444Z","iopub.status.idle":"2022-07-28T07:34:23.032356Z","shell.execute_reply.started":"2022-07-28T07:34:23.020419Z","shell.execute_reply":"2022-07-28T07:34:23.031315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc = SVC()\nsvc.fit(X_train, Y_train)\nY_pred = svc.predict(X_test)\nacc_svc = round(svc.score(X_train, Y_train)*100,2)\nacc_svc","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.039563Z","iopub.execute_input":"2022-07-28T07:34:23.039846Z","iopub.status.idle":"2022-07-28T07:34:23.120436Z","shell.execute_reply.started":"2022-07-28T07:34:23.039800Z","shell.execute_reply":"2022-07-28T07:34:23.119514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors=3)\nknn.fit(X_train, Y_train)\nY_pred = knn.predict(X_test)\nacc_knn = round(knn.score(X_train, Y_train)*100, 2)\nacc_knn","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.121601Z","iopub.execute_input":"2022-07-28T07:34:23.121873Z","iopub.status.idle":"2022-07-28T07:34:23.173584Z","shell.execute_reply.started":"2022-07-28T07:34:23.121841Z","shell.execute_reply":"2022-07-28T07:34:23.172833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gaussian = GaussianNB()\ngaussian.fit(X_train, Y_train)\nY_pred = gaussian.predict(X_test)\nacc_gaussian = round(gaussian.score(X_train, Y_train)*100, 2)\nacc_gaussian","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.174801Z","iopub.execute_input":"2022-07-28T07:34:23.175289Z","iopub.status.idle":"2022-07-28T07:34:23.188194Z","shell.execute_reply.started":"2022-07-28T07:34:23.175250Z","shell.execute_reply":"2022-07-28T07:34:23.187461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perceptron = Perceptron()\nperceptron.fit(X_train, Y_train)\nY_pred = perceptron.predict(X_test)\nacc_perceptron = round(perceptron.score(X_train, Y_train)*100, 2)\nacc_perceptron","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.189258Z","iopub.execute_input":"2022-07-28T07:34:23.189556Z","iopub.status.idle":"2022-07-28T07:34:23.203620Z","shell.execute_reply.started":"2022-07-28T07:34:23.189523Z","shell.execute_reply":"2022-07-28T07:34:23.202892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"linear_svc = LinearSVC()\nlinear_svc.fit(X_train, Y_train)\nY_pred = linear_svc.predict(X_test)\nacc_linear_svc = round(linear_svc.score(X_train, Y_train)*100, 2)\nacc_linear_svc","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.204808Z","iopub.execute_input":"2022-07-28T07:34:23.205412Z","iopub.status.idle":"2022-07-28T07:34:23.258467Z","shell.execute_reply.started":"2022-07-28T07:34:23.205375Z","shell.execute_reply":"2022-07-28T07:34:23.257450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sgd = SGDClassifier()\nsgd.fit(X_train, Y_train)\nY_pred = sgd.predict(X_test)\nacc_sgd = round(sgd.score(X_train, Y_train)*100, 2)\nacc_sgd","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.259700Z","iopub.execute_input":"2022-07-28T07:34:23.259968Z","iopub.status.idle":"2022-07-28T07:34:23.271012Z","shell.execute_reply.started":"2022-07-28T07:34:23.259943Z","shell.execute_reply":"2022-07-28T07:34:23.270103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, Y_train)\nY_pred = decision_tree.predict(X_test)\nacc_decison_tree = round(decision_tree.score(X_train, Y_train)*100, 2)\nacc_decison_tree","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.271876Z","iopub.execute_input":"2022-07-28T07:34:23.272493Z","iopub.status.idle":"2022-07-28T07:34:23.285986Z","shell.execute_reply.started":"2022-07-28T07:34:23.272468Z","shell.execute_reply":"2022-07-28T07:34:23.285074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_forest = RandomForestClassifier(n_estimators=100)\nrandom_forest.fit(X_train, Y_train)\nY_pred = random_forest.predict(X_test)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_train, Y_train)*100, 2)\nacc_random_forest","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.287303Z","iopub.execute_input":"2022-07-28T07:34:23.287872Z","iopub.status.idle":"2022-07-28T07:34:23.503930Z","shell.execute_reply.started":"2022-07-28T07:34:23.287839Z","shell.execute_reply":"2022-07-28T07:34:23.502956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame(\n    {\n        'Model' : [\n            'Support Vector Machines',\n            'KNN',\n            'Logistic Regression',\n            'Random Forest',\n            'Naive Bayes',\n            'Perceptron',\n            'Stochastic Gradient Descent',\n            'Linear SVC',\n            'Decision Tree'\n        ],\n        'Score' : [\n           acc_svc, acc_knn, acc_log, acc_random_forest, acc_gaussian, acc_perceptron, acc_sgd, acc_linear_svc, acc_decison_tree\n        ]\n    }\n)\nmodels.sort_values(by='Score', ascending=False)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.505038Z","iopub.execute_input":"2022-07-28T07:34:23.505328Z","iopub.status.idle":"2022-07-28T07:34:23.517011Z","shell.execute_reply.started":"2022-07-28T07:34:23.505301Z","shell.execute_reply":"2022-07-28T07:34:23.516098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(\n    {\n        \"PassengerId\" : test_df[\"PassengerId\"],\n        \"Survived\" : Y_pred\n    }\n)\nsubmission","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:34:23.518236Z","iopub.execute_input":"2022-07-28T07:34:23.518498Z","iopub.status.idle":"2022-07-28T07:34:23.530533Z","shell.execute_reply.started":"2022-07-28T07:34:23.518473Z","shell.execute_reply":"2022-07-28T07:34:23.529633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission.to_csv(cwd+SUBMISSION_PATH, header=True, index=False)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-28T07:35:14.932967Z","iopub.execute_input":"2022-07-28T07:35:14.933948Z","iopub.status.idle":"2022-07-28T07:35:14.937484Z","shell.execute_reply.started":"2022-07-28T07:35:14.933913Z","shell.execute_reply":"2022-07-28T07:35:14.936808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false}},"execution_count":null,"outputs":[]}]}