{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-08T01:55:28.952536Z","iopub.execute_input":"2022-08-08T01:55:28.953144Z","iopub.status.idle":"2022-08-08T01:55:28.966288Z","shell.execute_reply.started":"2022-08-08T01:55:28.953109Z","shell.execute_reply":"2022-08-08T01:55:28.964947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Workflow stages\n# \n\n1. Question or problem definition.\n2. Acquire training and testing data.\n3. Wrangle, prepare, cleanse the data.\n4. Analyze, identify patterns, and explore the data.\n5. Model, predict and solve the problem.\n6. Visualize, report, and present the problem solving steps and final solution.\n7. Supply or submit the results.***\n\n# Question and problem definition\nThe competition is simple: use machine learning to create a model that predicts which passengers survived the Titanic shipwreck.\nFor more details :\nhttps://www.kaggle.com/c/titanic\n","metadata":{}},{"cell_type":"code","source":"# Imports\n\n# pandas\nimport pandas as pd\nfrom pandas import Series,DataFrame\n\n# numpy, matplotlib, seaborn\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# machine learning models \nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.225340Z","iopub.execute_input":"2022-08-08T01:56:02.226043Z","iopub.status.idle":"2022-08-08T01:56:02.235407Z","shell.execute_reply.started":"2022-08-08T01:56:02.225992Z","shell.execute_reply":"2022-08-08T01:56:02.234035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting the datat","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/titanic/train.csv')\ntest_df = pd.read_csv('../input/titanic/test.csv')\ncombine = [train_df, test_df]\n# preview the data\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.237999Z","iopub.execute_input":"2022-08-08T01:56:02.238977Z","iopub.status.idle":"2022-08-08T01:56:02.275376Z","shell.execute_reply.started":"2022-08-08T01:56:02.238932Z","shell.execute_reply":"2022-08-08T01:56:02.274355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.276565Z","iopub.execute_input":"2022-08-08T01:56:02.277459Z","iopub.status.idle":"2022-08-08T01:56:02.295825Z","shell.execute_reply.started":"2022-08-08T01:56:02.277410Z","shell.execute_reply":"2022-08-08T01:56:02.294398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Selection:\nWe need now to do feature selection to the given data that will inhance the machine learning score .  We have two types of data categorical and numerical \nNow let's start .","metadata":{}},{"cell_type":"code","source":"#Data Tyde\ntrain_df.info()\nprint('--'*40)\ntest_df.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.297838Z","iopub.execute_input":"2022-08-08T01:56:02.298786Z","iopub.status.idle":"2022-08-08T01:56:02.326028Z","shell.execute_reply.started":"2022-08-08T01:56:02.298734Z","shell.execute_reply":"2022-08-08T01:56:02.324639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Numerical Data:\n1. PassengerId (int)\n2. SibSp (int)\n3. Parch (int)\n4. Age (float)\n5. Fare (float)\n\n# Categorical Data:\n1. Pclass (Ordinal)\n2. Name (Nominal)\n3. Embarked (Nominal)\n\n# Mixed Data\n1. Ticket\n2. Cabin \n\n","metadata":{}},{"cell_type":"markdown","source":"# Null Values \n\nCabin > Age > Embarked ( null values in the training dataset)\n\nCabin > Age are incomplete in( test dataset).","metadata":{}},{"cell_type":"code","source":"#Missed Values\n\ntrain_df.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.330551Z","iopub.execute_input":"2022-08-08T01:56:02.330999Z","iopub.status.idle":"2022-08-08T01:56:02.342508Z","shell.execute_reply.started":"2022-08-08T01:56:02.330962Z","shell.execute_reply":"2022-08-08T01:56:02.341086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.344149Z","iopub.execute_input":"2022-08-08T01:56:02.344939Z","iopub.status.idle":"2022-08-08T01:56:02.359322Z","shell.execute_reply.started":"2022-08-08T01:56:02.344889Z","shell.execute_reply":"2022-08-08T01:56:02.357983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.361233Z","iopub.execute_input":"2022-08-08T01:56:02.362136Z","iopub.status.idle":"2022-08-08T01:56:02.401893Z","shell.execute_reply.started":"2022-08-08T01:56:02.362084Z","shell.execute_reply":"2022-08-08T01:56:02.400776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### It means that :\n\n* Most passengers (> 75%) did not travel with parents or children.\n* 50% of the passengers didn't survive\n* The mean age of passengers is 30 and you will see the 25% is 28 and the 75% is 38\n* Nearly 50% of the passengers had no  siblings and/or spouse aboard.\n","metadata":{}},{"cell_type":"code","source":"train_df.describe(include=['O'])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.404294Z","iopub.execute_input":"2022-08-08T01:56:02.404707Z","iopub.status.idle":"2022-08-08T01:56:02.434889Z","shell.execute_reply.started":"2022-08-08T01:56:02.404665Z","shell.execute_reply":"2022-08-08T01:56:02.433651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# It means :\n\n* We have high ratio (22%) duplicates values of Tickets as the count is 891 and the unique is 681 \n* Cabin also has high ratio of dublicates values \n* As a result , we will drop them and becouse of the high precent of null values. \n","metadata":{}},{"cell_type":"code","source":"print(\"Before\", train_df.shape, test_df.shape, combine[0].shape, combine[1].shape)\n\ntrain_df = train_df.drop(['Ticket', 'Cabin'], axis=1)\ntest_df = test_df.drop(['Ticket', 'Cabin'], axis=1)\ncombine = [train_df, test_df]\n\n\"After\", train_df.shape, test_df.shape, combine[0].shape, combine[1].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.437033Z","iopub.execute_input":"2022-08-08T01:56:02.437525Z","iopub.status.idle":"2022-08-08T01:56:02.450885Z","shell.execute_reply.started":"2022-08-08T01:56:02.437475Z","shell.execute_reply":"2022-08-08T01:56:02.449570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"># According to the null values:\nWe fill the null values to  the Embarked and the Age \n ","metadata":{}},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Sex'] = dataset['Sex'].map( {'female': 1, 'male': 0} ).astype(int)\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.538219Z","iopub.execute_input":"2022-08-08T01:56:02.539409Z","iopub.status.idle":"2022-08-08T01:56:02.559180Z","shell.execute_reply.started":"2022-08-08T01:56:02.539346Z","shell.execute_reply":"2022-08-08T01:56:02.558187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"guess_ages = np.zeros((2,3))\nguess_ages","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.561205Z","iopub.execute_input":"2022-08-08T01:56:02.561550Z","iopub.status.idle":"2022-08-08T01:56:02.572752Z","shell.execute_reply.started":"2022-08-08T01:56:02.561505Z","shell.execute_reply":"2022-08-08T01:56:02.571652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    for i in range(0, 2):\n        for j in range(0, 3):\n            guess_df = dataset[(dataset['Sex'] == i) & \\\n                                  (dataset['Pclass'] == j+1)]['Age'].dropna()\n\n            # age_mean = guess_df.mean()\n            # age_std = guess_df.std()\n            # age_guess = rnd.uniform(age_mean - age_std, age_mean + age_std)\n\n            age_guess = guess_df.median()\n\n            # Convert random age float to nearest .5 age\n            guess_ages[i,j] = int( age_guess/0.5 + 0.5 ) * 0.5\n            \n    for i in range(0, 2):\n        for j in range(0, 3):\n            dataset.loc[ (dataset.Age.isnull()) & (dataset.Sex == i) & (dataset.Pclass == j+1),\\\n                    'Age'] = guess_ages[i,j]\n\n    dataset['Age'] = dataset['Age'].astype(int)\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:02.599645Z","iopub.execute_input":"2022-08-08T01:56:02.600035Z","iopub.status.idle":"2022-08-08T01:56:02.654881Z","shell.execute_reply.started":"2022-08-08T01:56:02.600004Z","shell.execute_reply":"2022-08-08T01:56:02.653627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill missed embarked values:  \ntrain_df.Embarked = train_df.Embarked.fillna(train_df.Embarked.dropna().max())\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:03.179076Z","iopub.execute_input":"2022-08-08T01:56:03.179489Z","iopub.status.idle":"2022-08-08T01:56:03.187057Z","shell.execute_reply.started":"2022-08-08T01:56:03.179454Z","shell.execute_reply":"2022-08-08T01:56:03.185732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['Fare'].fillna(test_df['Fare'].dropna().median(), inplace=True)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:03.189153Z","iopub.execute_input":"2022-08-08T01:56:03.189666Z","iopub.status.idle":"2022-08-08T01:56:03.210953Z","shell.execute_reply.started":"2022-08-08T01:56:03.189613Z","shell.execute_reply":"2022-08-08T01:56:03.209358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.corr()[\"Survived\"].sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:03.212938Z","iopub.execute_input":"2022-08-08T01:56:03.213447Z","iopub.status.idle":"2022-08-08T01:56:03.226976Z","shell.execute_reply.started":"2022-08-08T01:56:03.213385Z","shell.execute_reply":"2022-08-08T01:56:03.225715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc = {'figure.figsize':(10,6)})\nsns.heatmap(train_df.corr(), annot = True, fmt='.2g',cmap= 'YlGnBu')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:03.301418Z","iopub.execute_input":"2022-08-08T01:56:03.301875Z","iopub.status.idle":"2022-08-08T01:56:04.105705Z","shell.execute_reply.started":"2022-08-08T01:56:03.301836Z","shell.execute_reply":"2022-08-08T01:56:04.104359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlation\n\nWe will see The Correlation between Survived and other features such as Sex , fare and so on. We will see now the importanat of Data Engineering.","metadata":{}},{"cell_type":"code","source":"train_df['AgeBand'] = pd.cut(train_df['Age'], 5)\ntrain_df[['AgeBand', 'Survived']].groupby(['AgeBand'], as_index=False).mean().sort_values(by='AgeBand', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:04.108255Z","iopub.execute_input":"2022-08-08T01:56:04.109437Z","iopub.status.idle":"2022-08-08T01:56:04.132837Z","shell.execute_reply.started":"2022-08-08T01:56:04.109374Z","shell.execute_reply":"2022-08-08T01:56:04.131346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:    \n    dataset.loc[ dataset['Age'] <= 16, 'Age'] = 0\n    dataset.loc[(dataset['Age'] > 16) & (dataset['Age'] <= 32), 'Age'] = 1\n    dataset.loc[(dataset['Age'] > 32) & (dataset['Age'] <= 48), 'Age'] = 2\n    dataset.loc[(dataset['Age'] > 48) & (dataset['Age'] <= 64), 'Age'] = 3\n    dataset.loc[ dataset['Age'] > 64, 'Age']\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:04.134834Z","iopub.execute_input":"2022-08-08T01:56:04.135356Z","iopub.status.idle":"2022-08-08T01:56:04.166231Z","shell.execute_reply.started":"2022-08-08T01:56:04.135305Z","shell.execute_reply":"2022-08-08T01:56:04.164515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(['AgeBand'], axis=1)\ncombine = [train_df, test_df]\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:04.169111Z","iopub.execute_input":"2022-08-08T01:56:04.169696Z","iopub.status.idle":"2022-08-08T01:56:04.188116Z","shell.execute_reply.started":"2022-08-08T01:56:04.169643Z","shell.execute_reply":"2022-08-08T01:56:04.186677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].map( {'S': 0, 'C': 1, 'Q': 2} ).astype(int)\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:05.918814Z","iopub.execute_input":"2022-08-08T01:56:05.919220Z","iopub.status.idle":"2022-08-08T01:56:05.939930Z","shell.execute_reply.started":"2022-08-08T01:56:05.919186Z","shell.execute_reply":"2022-08-08T01:56:05.938733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['FareBand'] = pd.qcut(train_df['Fare'], 4)\ntrain_df[['FareBand', 'Survived']].groupby(['FareBand'], as_index=False).mean().sort_values(by='FareBand', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:06.508957Z","iopub.execute_input":"2022-08-08T01:56:06.509356Z","iopub.status.idle":"2022-08-08T01:56:06.533947Z","shell.execute_reply.started":"2022-08-08T01:56:06.509324Z","shell.execute_reply":"2022-08-08T01:56:06.532541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset.loc[ dataset['Fare'] <= 7.91, 'Fare'] = 0\n    dataset.loc[(dataset['Fare'] > 7.91) & (dataset['Fare'] <= 14.454), 'Fare'] = 1\n    dataset.loc[(dataset['Fare'] > 14.454) & (dataset['Fare'] <= 31), 'Fare']   = 2\n    dataset.loc[ dataset['Fare'] > 31, 'Fare'] = 3\n    dataset['Fare'] = dataset['Fare'].astype(int)\n\ntrain_df = train_df.drop(['FareBand'], axis=1)\ncombine = [train_df, test_df]\n    \ntrain_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:06.942840Z","iopub.execute_input":"2022-08-08T01:56:06.943273Z","iopub.status.idle":"2022-08-08T01:56:06.974203Z","shell.execute_reply.started":"2022-08-08T01:56:06.943239Z","shell.execute_reply":"2022-08-08T01:56:06.972900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['FamilySize'] = dataset['SibSp'] + dataset['Parch'] + 1\ntrain_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:06.976464Z","iopub.execute_input":"2022-08-08T01:56:06.976890Z","iopub.status.idle":"2022-08-08T01:56:06.996937Z","shell.execute_reply.started":"2022-08-08T01:56:06.976851Z","shell.execute_reply":"2022-08-08T01:56:06.995690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['IsAlone'] = 0\n    dataset.loc[dataset['FamilySize'] == 1, 'IsAlone'] = 1\ntrain_df = train_df.drop(['Parch', 'SibSp'], axis=1)\ntest_df = test_df.drop(['Parch', 'SibSp'], axis=1)\ncombine = [train_df, test_df]\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:06.998843Z","iopub.execute_input":"2022-08-08T01:56:06.999255Z","iopub.status.idle":"2022-08-08T01:56:07.024387Z","shell.execute_reply.started":"2022-08-08T01:56:06.999203Z","shell.execute_reply":"2022-08-08T01:56:07.023032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfor dataset in combine:\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train_df['Title'], train_df['Sex'])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:07.404476Z","iopub.execute_input":"2022-08-08T01:56:07.405797Z","iopub.status.idle":"2022-08-08T01:56:07.436269Z","shell.execute_reply.started":"2022-08-08T01:56:07.405739Z","shell.execute_reply":"2022-08-08T01:56:07.434660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Countess','Capt', 'Col',\\\n \t'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n    \ntrain_df[['Title', 'Survived']].groupby(['Title'], as_index=False).mean()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:07.445874Z","iopub.execute_input":"2022-08-08T01:56:07.446306Z","iopub.status.idle":"2022-08-08T01:56:07.474258Z","shell.execute_reply.started":"2022-08-08T01:56:07.446271Z","shell.execute_reply":"2022-08-08T01:56:07.472976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"title_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].map(title_mapping)\n    dataset['Title'] = dataset['Title'].fillna(0)\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:07.476311Z","iopub.execute_input":"2022-08-08T01:56:07.476677Z","iopub.status.idle":"2022-08-08T01:56:07.498747Z","shell.execute_reply.started":"2022-08-08T01:56:07.476636Z","shell.execute_reply":"2022-08-08T01:56:07.497792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.corr()[\"Survived\"].sort_values(ascending=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:07.500198Z","iopub.execute_input":"2022-08-08T01:56:07.501176Z","iopub.status.idle":"2022-08-08T01:56:07.511352Z","shell.execute_reply.started":"2022-08-08T01:56:07.501134Z","shell.execute_reply":"2022-08-08T01:56:07.510353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# New_Correlation\n\n# Before \n\n1. Survived       1.000000\n2. Sex            0.543351\n3. Fare           0.257307\n4. Parch          0.081629\n5. SibSp         -0.035322\n6. Age           -0.060291\n7. Pclass        -0.338481\n. Name: Survived, dtype: float64\n \n# After\n\n1. Survived       1.000000\n2. Sex            0.543351\n3. Title          0.407753\n4. Fare           0.295875\n5. Embarked       0.106811\n6. Age           -0.060291\n7. IsAlone       -0.203367\n8. Pclass        -0.338481\n Name: Survived, dtype: float64 ","metadata":{}},{"cell_type":"code","source":"sns.set(rc = {'figure.figsize':(10,6)})\nsns.heatmap(train_df.corr(), annot = True, fmt='.2g',cmap= 'YlGnBu')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:07.527989Z","iopub.execute_input":"2022-08-08T01:56:07.528679Z","iopub.status.idle":"2022-08-08T01:56:08.308654Z","shell.execute_reply.started":"2022-08-08T01:56:07.528625Z","shell.execute_reply":"2022-08-08T01:56:08.307748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(['Name', 'PassengerId'], axis=1)\ntest_df = test_df.drop(['Name'], axis=1)\ncombine = [train_df, test_df]\ntrain_df.shape, test_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:08.310240Z","iopub.execute_input":"2022-08-08T01:56:08.311430Z","iopub.status.idle":"2022-08-08T01:56:08.322534Z","shell.execute_reply.started":"2022-08-08T01:56:08.311387Z","shell.execute_reply":"2022-08-08T01:56:08.321449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:08.629500Z","iopub.execute_input":"2022-08-08T01:56:08.629973Z","iopub.status.idle":"2022-08-08T01:56:08.643994Z","shell.execute_reply.started":"2022-08-08T01:56:08.629934Z","shell.execute_reply":"2022-08-08T01:56:08.642943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ML models ","metadata":{}},{"cell_type":"code","source":"X_train = train_df.drop(\"Survived\", axis=1)\nY_train = train_df[\"Survived\"]\nX_test  = test_df.drop(\"PassengerId\", axis=1).copy()\nX_train.shape, Y_train.shape, X_test.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:08.852940Z","iopub.execute_input":"2022-08-08T01:56:08.853557Z","iopub.status.idle":"2022-08-08T01:56:08.864303Z","shell.execute_reply.started":"2022-08-08T01:56:08.853523Z","shell.execute_reply":"2022-08-08T01:56:08.862886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression\nlogreg = LogisticRegression()\nlogreg.fit(X_train, Y_train)\nY_pred = logreg.predict(X_test)\nacc_log = round(logreg.score(X_train, Y_train) * 100, 2)\nacc_log","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:08.893413Z","iopub.execute_input":"2022-08-08T01:56:08.894548Z","iopub.status.idle":"2022-08-08T01:56:08.922600Z","shell.execute_reply.started":"2022-08-08T01:56:08.894501Z","shell.execute_reply":"2022-08-08T01:56:08.921678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Support Vector Machines\n\nsvc = SVC()\nsvc.fit(X_train, Y_train)\nY_pred = svc.predict(X_test)\nacc_svc = round(svc.score(X_train, Y_train) * 100, 2)\nacc_svc","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:08.924053Z","iopub.execute_input":"2022-08-08T01:56:08.925122Z","iopub.status.idle":"2022-08-08T01:56:08.995846Z","shell.execute_reply.started":"2022-08-08T01:56:08.925072Z","shell.execute_reply":"2022-08-08T01:56:08.994383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors = 3)\nknn.fit(X_train, Y_train)\nY_pred = knn.predict(X_test)\nacc_knn = round(knn.score(X_train, Y_train) * 100, 2)\nacc_knn","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:09.031326Z","iopub.execute_input":"2022-08-08T01:56:09.031783Z","iopub.status.idle":"2022-08-08T01:56:09.091790Z","shell.execute_reply.started":"2022-08-08T01:56:09.031746Z","shell.execute_reply":"2022-08-08T01:56:09.090507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gaussian = GaussianNB()\ngaussian.fit(X_train, Y_train)\nY_pred = gaussian.predict(X_test)\nacc_gaussian = round(gaussian.score(X_train, Y_train) * 100, 2)\nacc_gaussian","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:09.125344Z","iopub.execute_input":"2022-08-08T01:56:09.125781Z","iopub.status.idle":"2022-08-08T01:56:09.143539Z","shell.execute_reply.started":"2022-08-08T01:56:09.125742Z","shell.execute_reply":"2022-08-08T01:56:09.142678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perceptron\n\nperceptron = Perceptron()\nperceptron.fit(X_train, Y_train)\nY_pred = perceptron.predict(X_test)\nacc_perceptron = round(perceptron.score(X_train, Y_train) * 100, 2)\nacc_perceptron","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:09.145283Z","iopub.execute_input":"2022-08-08T01:56:09.145944Z","iopub.status.idle":"2022-08-08T01:56:09.160682Z","shell.execute_reply.started":"2022-08-08T01:56:09.145904Z","shell.execute_reply":"2022-08-08T01:56:09.159874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Linear SVC\n\nlinear_svc = LinearSVC()\nlinear_svc.fit(X_train, Y_train)\nY_pred = linear_svc.predict(X_test)\nacc_linear_svc = round(linear_svc.score(X_train, Y_train) * 100, 2)\nacc_linear_svc","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:20.324207Z","iopub.execute_input":"2022-08-08T01:56:20.324998Z","iopub.status.idle":"2022-08-08T01:56:20.389840Z","shell.execute_reply.started":"2022-08-08T01:56:20.324953Z","shell.execute_reply":"2022-08-08T01:56:20.388663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stochastic Gradient Descent\n\nsgd = SGDClassifier()\nsgd.fit(X_train, Y_train)\nY_pred = sgd.predict(X_test)\nacc_sgd = round(sgd.score(X_train, Y_train) * 100, 2)\nacc_sgd","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:25.381023Z","iopub.execute_input":"2022-08-08T01:56:25.381468Z","iopub.status.idle":"2022-08-08T01:56:25.400295Z","shell.execute_reply.started":"2022-08-08T01:56:25.381429Z","shell.execute_reply":"2022-08-08T01:56:25.398934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decision Tree\n\ndecision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, Y_train)\nY_pred = decision_tree.predict(X_test)\nacc_decision_tree = round(decision_tree.score(X_train, Y_train) * 100, 2)\nacc_decision_tree","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:26.213093Z","iopub.execute_input":"2022-08-08T01:56:26.214345Z","iopub.status.idle":"2022-08-08T01:56:26.232803Z","shell.execute_reply.started":"2022-08-08T01:56:26.214296Z","shell.execute_reply":"2022-08-08T01:56:26.231258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Forest\n\nrandom_forest = RandomForestClassifier(n_estimators=100)\nrandom_forest.fit(X_train, Y_train)\nY_pred = random_forest.predict(X_test)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_train, Y_train) * 100, 2)\nacc_random_forest","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:26.235390Z","iopub.execute_input":"2022-08-08T01:56:26.235945Z","iopub.status.idle":"2022-08-08T01:56:26.515184Z","shell.execute_reply.started":"2022-08-08T01:56:26.235890Z","shell.execute_reply":"2022-08-08T01:56:26.513897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Logistic Regression', \n              'Random Forest', 'Naive Bayes', 'Perceptron', \n              'Stochastic Gradient Decent', 'Linear SVC', \n              'Decision Tree'],\n    'Score': [acc_svc, acc_knn, acc_log, \n              acc_random_forest, acc_gaussian, acc_perceptron, \n              acc_sgd, acc_linear_svc, acc_decision_tree]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:27.030102Z","iopub.execute_input":"2022-08-08T01:56:27.031018Z","iopub.status.idle":"2022-08-08T01:56:27.047401Z","shell.execute_reply.started":"2022-08-08T01:56:27.030970Z","shell.execute_reply":"2022-08-08T01:56:27.046341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        \"PassengerId\": test_df[\"PassengerId\"],\n        \"Survived\": Y_pred\n    })\nsubmission.to_csv('titanic.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T01:56:27.162664Z","iopub.execute_input":"2022-08-08T01:56:27.163501Z","iopub.status.idle":"2022-08-08T01:56:27.173478Z","shell.execute_reply.started":"2022-08-08T01:56:27.163457Z","shell.execute_reply":"2022-08-08T01:56:27.172138Z"},"trusted":true},"execution_count":null,"outputs":[]}]}