{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# data analysis and wrangling\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n\n# machine learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.model_selection import cross_validate, GridSearchCV\nfrom sklearn.metrics import classification_report, accuracy_score, f1_score, confusion_matrix\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T10:06:31.905257Z","iopub.execute_input":"2022-07-21T10:06:31.905789Z","iopub.status.idle":"2022-07-21T10:06:31.923462Z","shell.execute_reply.started":"2022-07-21T10:06:31.905735Z","shell.execute_reply":"2022-07-21T10:06:31.922310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get Data\nAfter importing the required libarires, next step is to get the train and test data from csv to dataframe.","metadata":{}},{"cell_type":"code","source":"train=pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest= pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:31.925737Z","iopub.execute_input":"2022-07-21T10:06:31.926154Z","iopub.status.idle":"2022-07-21T10:06:31.955165Z","shell.execute_reply.started":"2022-07-21T10:06:31.926113Z","shell.execute_reply":"2022-07-21T10:06:31.954012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"let's check the basic information about train & test data","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:31.956868Z","iopub.execute_input":"2022-07-21T10:06:31.957950Z","iopub.status.idle":"2022-07-21T10:06:31.978435Z","shell.execute_reply.started":"2022-07-21T10:06:31.957895Z","shell.execute_reply":"2022-07-21T10:06:31.976729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:31.982422Z","iopub.execute_input":"2022-07-21T10:06:31.983996Z","iopub.status.idle":"2022-07-21T10:06:32.005746Z","shell.execute_reply.started":"2022-07-21T10:06:31.983944Z","shell.execute_reply":"2022-07-21T10:06:32.004670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.007402Z","iopub.execute_input":"2022-07-21T10:06:32.008551Z","iopub.status.idle":"2022-07-21T10:06:32.024736Z","shell.execute_reply.started":"2022-07-21T10:06:32.008511Z","shell.execute_reply":"2022-07-21T10:06:32.023488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.027254Z","iopub.execute_input":"2022-07-21T10:06:32.027786Z","iopub.status.idle":"2022-07-21T10:06:32.040352Z","shell.execute_reply.started":"2022-07-21T10:06:32.027737Z","shell.execute_reply":"2022-07-21T10:06:32.039419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Because Cabin column has a very large number of null values we are dropping it from both the data sets.","metadata":{}},{"cell_type":"markdown","source":"Drop Useless Columns [First level]","metadata":{}},{"cell_type":"code","source":"train.drop(\"Cabin\", axis=1,inplace=True)\ntest.drop(\"Cabin\", axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.041882Z","iopub.execute_input":"2022-07-21T10:06:32.042699Z","iopub.status.idle":"2022-07-21T10:06:32.053582Z","shell.execute_reply.started":"2022-07-21T10:06:32.042666Z","shell.execute_reply":"2022-07-21T10:06:32.052460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dealing with Null Values \nBy applying mean method for age coloum and median mode for embarked","metadata":{}},{"cell_type":"code","source":"train.Embarked.unique()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.055173Z","iopub.execute_input":"2022-07-21T10:06:32.056403Z","iopub.status.idle":"2022-07-21T10:06:32.066874Z","shell.execute_reply.started":"2022-07-21T10:06:32.056353Z","shell.execute_reply":"2022-07-21T10:06:32.065924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"moEm=train.Embarked.mode()\nmoEm","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.068171Z","iopub.execute_input":"2022-07-21T10:06:32.069006Z","iopub.status.idle":"2022-07-21T10:06:32.080203Z","shell.execute_reply.started":"2022-07-21T10:06:32.068967Z","shell.execute_reply":"2022-07-21T10:06:32.079263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Embarked.fillna(moEm[0], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.085201Z","iopub.execute_input":"2022-07-21T10:06:32.085926Z","iopub.status.idle":"2022-07-21T10:06:32.092711Z","shell.execute_reply.started":"2022-07-21T10:06:32.085886Z","shell.execute_reply":"2022-07-21T10:06:32.091648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we will guess the null values of age from Pclass and Sex:\nguess_ages = np.zeros((2,3))\nguess_ages\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.094844Z","iopub.execute_input":"2022-07-21T10:06:32.095759Z","iopub.status.idle":"2022-07-21T10:06:32.106728Z","shell.execute_reply.started":"2022-07-21T10:06:32.095708Z","shell.execute_reply":"2022-07-21T10:06:32.105745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we iterate over Sex (0 or 1) and Pclass (1, 2, 3) to calculate guessed values of Age for the six combinations.\n\ncombine = [train , test]\n\n# Converting Sex categories (male and female) to 0 and 1:\nfor dataset in combine:\n    dataset['Sex'] = dataset['Sex'].map( {'female': 1, 'male': 0} ).astype(int)\n\n# Filling missed age feature:\n\nfor dataset in combine:\n    for i in range(0, 2):\n        for j in range(0, 3):\n            guess_df = dataset[(dataset['Sex'] == i) & \\\n                                  (dataset['Pclass'] == j+1)]['Age'].dropna()\n            age_guess = guess_df.median()\n\n            # Convert random age float to nearest .5 age\n            guess_ages[i,j] = int( age_guess/0.5 + 0.5 ) * 0.5\n            \n    for i in range(0, 2):\n        for j in range(0, 3):\n            dataset.loc[ (dataset.Age.isnull()) & (dataset.Sex == i) & (dataset.Pclass == j+1),\\\n                    'Age'] = guess_ages[i,j]\n\n    dataset['Age'] = dataset['Age'].astype(int)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.108162Z","iopub.execute_input":"2022-07-21T10:06:32.109011Z","iopub.status.idle":"2022-07-21T10:06:32.169087Z","shell.execute_reply.started":"2022-07-21T10:06:32.108974Z","shell.execute_reply":"2022-07-21T10:06:32.168128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What is the distribution of numerical feature values across the samples?\n\nThis helps us determine, among other early insights, how representative is the training dataset of the actual problem domain.\n\nTotal samples are 891 or 40% of the actual number of passengers on board the Titanic (2,224).\nSurvived is a categorical feature with 0 or 1 values.\nAround 38% samples survived representative of the actual survival rate at 32%.\nMost passengers (> 75%) did not travel with parents or children.\nNearly 30% of the passengers had siblings and/or spouse aboard.\nFares varied significantly with few passengers (<1%) paying as high as $512.\nFew elderly passengers (<1%) within age range 65-80.","metadata":{}},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.170329Z","iopub.execute_input":"2022-07-21T10:06:32.171596Z","iopub.status.idle":"2022-07-21T10:06:32.214950Z","shell.execute_reply.started":"2022-07-21T10:06:32.171522Z","shell.execute_reply":"2022-07-21T10:06:32.213659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Assumtions based on data analysis\nWe arrive at following assumptions based on data analysis done so far. We may validate these assumptions further before taking appropriate actions.\n\nCreating.\n\n1.We may want to create a new feature called Family based on Parch and SibSp to get total count of family members on board.\n2.We may want to engineer the Name feature to extract Title as a new feature.\n3.We may want to create new feature for Age bands. This turns a continous numerical feature into an ordinal categorical feature.\n4.We may also want to create a Fare range feature if it helps our analysis.\n\n\nClassifying.\n\nWe may also add to our assumptions based on the problem description noted earlier.\n\n1.Women (Sex=female) were more likely to have survived.\n2.Children (Age<?) were more likely to have survived.\n3.The upper-class passengers (Pclass=1) were more likely to have survived.","metadata":{}},{"cell_type":"markdown","source":"# Analyze by pivoting features\nTo confirm some of our observations and assumptions, we can quickly analyze our feature correlations by pivoting features against each other. We can only do so at this stage for features which do not have any empty values. It also makes sense doing so only for features which are categorical (Sex), ordinal (Pclass) or discrete (SibSp, Parch) type.\n\nPclass We observe significant correlation (>0.5) among Pclass=1 and Survived (classifying #3). We decide to include this feature in our model.\nSex We confirm the observation during problem definition that Sex=female had very high survival rate at 74% (classifying #1).\nSibSp and Parch These features have zero correlation for certain values. It may be best to derive a feature or a set of features from these individual features (creating #1)","metadata":{}},{"cell_type":"code","source":"train[['Pclass','Survived']].groupby(['Pclass'],as_index=False).mean().sort_values(by='Survived',ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.217214Z","iopub.execute_input":"2022-07-21T10:06:32.217760Z","iopub.status.idle":"2022-07-21T10:06:32.236567Z","shell.execute_reply.started":"2022-07-21T10:06:32.217708Z","shell.execute_reply":"2022-07-21T10:06:32.235025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[[\"Sex\", \"Survived\"]].groupby(['Sex'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.238435Z","iopub.execute_input":"2022-07-21T10:06:32.239668Z","iopub.status.idle":"2022-07-21T10:06:32.256798Z","shell.execute_reply.started":"2022-07-21T10:06:32.239611Z","shell.execute_reply":"2022-07-21T10:06:32.255777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[[\"SibSp\", \"Survived\"]].groupby(['SibSp'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.258169Z","iopub.execute_input":"2022-07-21T10:06:32.259422Z","iopub.status.idle":"2022-07-21T10:06:32.282070Z","shell.execute_reply.started":"2022-07-21T10:06:32.259365Z","shell.execute_reply":"2022-07-21T10:06:32.281135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[[\"Parch\", \"Survived\"]].groupby(['Parch'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.283791Z","iopub.execute_input":"2022-07-21T10:06:32.284424Z","iopub.status.idle":"2022-07-21T10:06:32.301223Z","shell.execute_reply.started":"2022-07-21T10:06:32.284390Z","shell.execute_reply":"2022-07-21T10:06:32.300408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" drop useless coloums[Second Level] and saving it into different dataframe","metadata":{}},{"cell_type":"markdown","source":"# Analyze by visualizing data\nNow we can continue confirming some of our assumptions using visualizations for analyzing the data.","metadata":{}},{"cell_type":"markdown","source":"Observations.\n\nInfants (Age <=4) had high survival rate.\n\nOldest passengers (Age = 80) survived.\n\nLarge number of 15-25 year olds did not survive.\n\nMost passengers are in 15-35 age range.\n","metadata":{}},{"cell_type":"code","source":"g=sns.FacetGrid(train, col='Survived')\ng.map(plt.hist, 'Age', bins=20)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:32.302542Z","iopub.execute_input":"2022-07-21T10:06:32.302887Z","iopub.status.idle":"2022-07-21T10:06:33.438123Z","shell.execute_reply.started":"2022-07-21T10:06:32.302856Z","shell.execute_reply":"2022-07-21T10:06:33.436789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations.\n\nPclass=3 had most passengers, however most did not survive. Confirms our classifying assumption #2.\n\nInfant passengers in Pclass=2 and Pclass=3 mostly survived. Further qualifies our classifying assumption #2.\n\nMost passengers in Pclass=1 survived. Confirms our classifying assumption #3.\n\nPclass varies in terms of Age distribution of passengers","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:32:31.730652Z","iopub.execute_input":"2022-07-20T09:32:31.731024Z","iopub.status.idle":"2022-07-20T09:32:31.737913Z","shell.execute_reply.started":"2022-07-20T09:32:31.730993Z","shell.execute_reply":"2022-07-20T09:32:31.736956Z"}}},{"cell_type":"code","source":"grid = sns.FacetGrid(train, col='Survived', row='Pclass', size=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:33.439722Z","iopub.execute_input":"2022-07-21T10:06:33.440308Z","iopub.status.idle":"2022-07-21T10:06:34.853248Z","shell.execute_reply.started":"2022-07-21T10:06:33.440272Z","shell.execute_reply":"2022-07-21T10:06:34.852015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations.\n\nFemale passengers had much better survival rate than males. Confirms classifying (#1).\n\nException in Embarked=C where males had higher survival rate. This could be a correlation between Pclass and Embarked and in turn Pclass and Survived, not necessarily direct correlation between Embarked and Survived.\n\nMales had better survival rate in Pclass=3 when compared with Pclass=2 for C and Q ports. Completing (#2).\n\nPorts of embarkation have varying survival rates for Pclass=3 and among male passengers. Correlating (#1).","metadata":{}},{"cell_type":"code","source":"grid = sns.FacetGrid(train, row='Embarked', size=2.2, aspect=1.6)\ngrid.map(sns.pointplot, 'Pclass', 'Survived', 'Sex', palette='deep')\ngrid.add_legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:34.855100Z","iopub.execute_input":"2022-07-21T10:06:34.855784Z","iopub.status.idle":"2022-07-21T10:06:36.141931Z","shell.execute_reply.started":"2022-07-21T10:06:34.855731Z","shell.execute_reply":"2022-07-21T10:06:36.140453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Wrangle data\nWe have collected several assumptions and decisions regarding our datasets and solution requirements. So far we did not have to change a single feature or value to arrive at these. Let us now execute our decisions and assumptions for correcting, creating, and completing goals.","metadata":{}},{"cell_type":"markdown","source":" dropping features\n \nThis is a good starting goal to execute. By dropping features we are dealing with fewer data points. Speeds up our notebook and eases the analysis.\n\nBased on our assumptions and decisions we want to drop the Ticket  features.\n\nNote that where applicable we perform operations on both training and testing datasets together to stay consistent.","metadata":{}},{"cell_type":"code","source":"train = train.drop(['Ticket'], axis=1)\ntest= test.drop(['Ticket'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.144264Z","iopub.execute_input":"2022-07-21T10:06:36.144763Z","iopub.status.idle":"2022-07-21T10:06:36.154216Z","shell.execute_reply.started":"2022-07-21T10:06:36.144721Z","shell.execute_reply":"2022-07-21T10:06:36.152192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creating new feature extracting from existing\nWe want to analyze if Name feature can be engineered to extract titles and test correlation between titles and survival, before dropping Name and PassengerId features.\n\nIn the following code we extract Title feature using regular expressions. The RegEx pattern (\\w+\\.) matches the first word which ends with a dot character within Name feature. The expand=False flag returns a DataFrame.\n\nObservations.\n\nWhen we plot Title, Age, and Survived, we note the following observations.\n\nMost titles band Age groups accurately. For example: Master title has Age mean of 5 years.\n\nSurvival among Title Age bands varies slightly.\n\nCertain titles mostly survived (Mme, Lady, Sir) or did not (Don, Rev, Jonkheer).\n\nDecision.\n\nWe decide to retain the new Title feature for model training.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"combine = [train, test]\n\nfor dataset in combine:\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train['Title'], train['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.156396Z","iopub.execute_input":"2022-07-21T10:06:36.157763Z","iopub.status.idle":"2022-07-21T10:06:36.195500Z","shell.execute_reply.started":"2022-07-21T10:06:36.157673Z","shell.execute_reply":"2022-07-21T10:06:36.194028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can replace many titles with a more common name or classify them as Rare.\n\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Countess','Capt', 'Col',\\\n \t'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n    \ntrain[['Title', 'Survived']].groupby(['Title'], as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.197761Z","iopub.execute_input":"2022-07-21T10:06:36.198371Z","iopub.status.idle":"2022-07-21T10:06:36.226799Z","shell.execute_reply.started":"2022-07-21T10:06:36.198316Z","shell.execute_reply":"2022-07-21T10:06:36.225736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can convert the categorical titles to ordinal.\n\ntitle_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].map(title_mapping)\n    dataset['Title'] = dataset['Title'].fillna(0)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.228689Z","iopub.execute_input":"2022-07-21T10:06:36.229043Z","iopub.status.idle":"2022-07-21T10:06:36.252540Z","shell.execute_reply.started":"2022-07-21T10:06:36.229009Z","shell.execute_reply":"2022-07-21T10:06:36.251315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we can safely drop the Name feature from training and testing datasets. We also do not need the PassengerId feature in the training dataset.","metadata":{}},{"cell_type":"code","source":"# train = train.drop(['Name', 'PassengerId'], axis=1)\ntest = test.drop(['Name'], axis=1)\ncombine = [train, test]\ntrain.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.254305Z","iopub.execute_input":"2022-07-21T10:06:36.254771Z","iopub.status.idle":"2022-07-21T10:06:36.267867Z","shell.execute_reply.started":"2022-07-21T10:06:36.254690Z","shell.execute_reply":"2022-07-21T10:06:36.266354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Converting a categorical feature\nNow we can convert features which contain strings to numerical values. This is required by most model algorithms. Doing so will also help us in achieving the feature completing goal.\n\nLet us start by converting Sex feature to a new feature called Gender where female=1 and male=0.","metadata":{}},{"cell_type":"markdown","source":"Let us create Age bands and determine correlations with Survived.","metadata":{}},{"cell_type":"markdown","source":"Let us replace Age with ordinals based on these bands.","metadata":{}},{"cell_type":"markdown","source":"Create new feature combining existing features\n\nWe can create a new feature for FamilySize which combines Parch and SibSp. This will enable us to drop Parch and SibSp from our datasets.","metadata":{}},{"cell_type":"code","source":"for dataset in combine:\n    dataset['FamilySize'] = dataset['SibSp'] + dataset['Parch'] + 1\n\ntrain[['FamilySize', 'Survived']].groupby(['FamilySize'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.269334Z","iopub.execute_input":"2022-07-21T10:06:36.270054Z","iopub.status.idle":"2022-07-21T10:06:36.295827Z","shell.execute_reply.started":"2022-07-21T10:06:36.270013Z","shell.execute_reply":"2022-07-21T10:06:36.294766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can create another feature called IsAlone.\n\nfor dataset in combine:\n    dataset['IsAlone'] = 0\n    dataset.loc[dataset['FamilySize'] == 1, 'IsAlone'] = 1\n\ntrain[['IsAlone', 'Survived']].groupby(['IsAlone'], as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.297134Z","iopub.execute_input":"2022-07-21T10:06:36.297609Z","iopub.status.idle":"2022-07-21T10:06:36.315701Z","shell.execute_reply.started":"2022-07-21T10:06:36.297539Z","shell.execute_reply":"2022-07-21T10:06:36.314524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us drop Parch, SibSp, and FamilySize features in favor of IsAlone.\n\ntrain= train.drop(['Parch', 'SibSp','Name'], axis=1)\ntest = test.drop(['Parch', 'SibSp'], axis=1)\ncombine = [train, test]\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.323156Z","iopub.execute_input":"2022-07-21T10:06:36.324383Z","iopub.status.idle":"2022-07-21T10:06:36.343971Z","shell.execute_reply.started":"2022-07-21T10:06:36.324333Z","shell.execute_reply":"2022-07-21T10:06:36.342648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Age*Class'] = dataset.Age * dataset.Pclass\n\ntrain.loc[:, ['Age*Class', 'Age', 'Pclass']].head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.346105Z","iopub.execute_input":"2022-07-21T10:06:36.346569Z","iopub.status.idle":"2022-07-21T10:06:36.361940Z","shell.execute_reply.started":"2022-07-21T10:06:36.346509Z","shell.execute_reply":"2022-07-21T10:06:36.360991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Dataset\ntest.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.363385Z","iopub.execute_input":"2022-07-21T10:06:36.364651Z","iopub.status.idle":"2022-07-21T10:06:36.385253Z","shell.execute_reply.started":"2022-07-21T10:06:36.364617Z","shell.execute_reply":"2022-07-21T10:06:36.383650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Convert Categorical cols to Numerical cols","metadata":{}},{"cell_type":"code","source":"train=pd.get_dummies(train, columns=[\"Embarked\"])\ntest=pd.get_dummies(test, columns=[\"Embarked\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.387888Z","iopub.execute_input":"2022-07-21T10:06:36.389088Z","iopub.status.idle":"2022-07-21T10:06:36.403137Z","shell.execute_reply.started":"2022-07-21T10:06:36.389021Z","shell.execute_reply":"2022-07-21T10:06:36.401867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['Embarked_C'],axis=1,inplace=True)\ntest.drop(['Embarked_C'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.405248Z","iopub.execute_input":"2022-07-21T10:06:36.406152Z","iopub.status.idle":"2022-07-21T10:06:36.418495Z","shell.execute_reply.started":"2022-07-21T10:06:36.406100Z","shell.execute_reply":"2022-07-21T10:06:36.417303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling null values of Fare by taking mean of this Class\ntest.Fare.fillna(train.groupby(\"Pclass\").mean()[\"Fare\"][3], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.420811Z","iopub.execute_input":"2022-07-21T10:06:36.421817Z","iopub.status.idle":"2022-07-21T10:06:36.439609Z","shell.execute_reply.started":"2022-07-21T10:06:36.421763Z","shell.execute_reply":"2022-07-21T10:06:36.438223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Scale the data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\nscaler=MinMaxScaler()\ntrain['Pclass'] = scaler.fit_transform(train[['Pclass']])\ntrain['Age']=scaler.fit_transform(train[['Age']])\ntrain['Fare']=scaler.fit_transform(train[['Fare']])\ntest['Pclass'] = scaler.fit_transform(test[['Pclass']])\ntest['Age']=scaler.fit_transform(test[['Age']])\ntest['Fare']=scaler.fit_transform(test[['Fare']])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.441802Z","iopub.execute_input":"2022-07-21T10:06:36.442678Z","iopub.status.idle":"2022-07-21T10:06:36.475514Z","shell.execute_reply.started":"2022-07-21T10:06:36.442631Z","shell.execute_reply":"2022-07-21T10:06:36.474178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.477294Z","iopub.execute_input":"2022-07-21T10:06:36.478324Z","iopub.status.idle":"2022-07-21T10:06:36.497489Z","shell.execute_reply.started":"2022-07-21T10:06:36.478280Z","shell.execute_reply":"2022-07-21T10:06:36.496106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model, predict and solve\nNow we are ready to train a model and predict the required solution. There are 60+ predictive modelling algorithms to choose from. We must understand the type of problem and solution requirement to narrow down to a select few models which we can evaluate. Our problem is a classification and regression problem. We want to identify relationship between output (Survived or not) with other variables or features (Gender, Age, Port...). We are also perfoming a category of machine learning which is called supervised learning as we are training our model with a given dataset. With these two criteria - Supervised Learning plus Classification and Regression, we can narrow down our choice of models to a few. These include:\n\nLogistic Regression\nKNN or k-Nearest Neighbors\nSupport Vector Machines\nDecision Tree\nRandom Forrest\nArtificial neural network\nRVM or Relevance Vector Machine","metadata":{}},{"cell_type":"code","source":"X_train = train.drop(\"Survived\", axis=1)\nY_train = train[\"Survived\"]\nX_test  = test.copy()\nX_train.shape, Y_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.499627Z","iopub.execute_input":"2022-07-21T10:06:36.500157Z","iopub.status.idle":"2022-07-21T10:06:36.511988Z","shell.execute_reply.started":"2022-07-21T10:06:36.500107Z","shell.execute_reply":"2022-07-21T10:06:36.510742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameter Tuning","metadata":{}},{"cell_type":"markdown","source":"Best hyperparameter: {'C': 1.0, 'penalty': 'l2', 'tol': 0.1}","metadata":{}},{"cell_type":"markdown","source":"Split data into features and target","metadata":{}},{"cell_type":"markdown","source":"# Logistic Regression\n\nlogreg = LogisticRegression()\nlogreg.fit(X_train, Y_train)\nY_pred_log = logreg.predict(X_test)\nacc_log = round(logreg.score(X_train, Y_train) * 100, 2)\nacc_log","metadata":{"execution":{"iopub.status.busy":"2022-07-21T09:08:07.410131Z","iopub.execute_input":"2022-07-21T09:08:07.411433Z","iopub.status.idle":"2022-07-21T09:08:07.498757Z","shell.execute_reply.started":"2022-07-21T09:08:07.411394Z","shell.execute_reply":"2022-07-21T09:08:07.497613Z"}}},{"cell_type":"markdown","source":"We can use Logistic Regression to validate our assumptions and decisions for feature creating and completing goals. This can be done by calculating the coefficient of the features in the decision function.\n\nPositive coefficients increase the log-odds of the response (and thus increase the probability), and negative coefficients decrease the log-odds of the response (and thus decrease the probability).\n\nSex is highest positivie coefficient, implying as the Sex value increases (male: 0 to female: 1), the probability of Survived=1 increases the most.\nInversely as Pclass increases, probability of Survived=1 decreases the most.\nThis way Age*Class is a good artificial feature to model as it has second highest negative correlation with Survived.\nSo is Title as second highest positive correlation.","metadata":{}},{"cell_type":"code","source":"# Logistic Regression\n\nlogreg = LogisticRegression()\nlogreg.fit(X_train, Y_train)\nY_pred = logreg.predict(X_test)\nacc_log = round(logreg.score(X_train, Y_train) * 100, 2)\nacc_log","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.513961Z","iopub.execute_input":"2022-07-21T10:06:36.514372Z","iopub.status.idle":"2022-07-21T10:06:36.597890Z","shell.execute_reply.started":"2022-07-21T10:06:36.514330Z","shell.execute_reply":"2022-07-21T10:06:36.596232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coeff_df = pd.DataFrame(train.columns.delete(0))\ncoeff_df.columns = ['Feature']\ncoeff_df[\"Correlation\"] = pd.Series(logreg.coef_[0])\n\ncoeff_df.sort_values(by='Correlation', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.600535Z","iopub.execute_input":"2022-07-21T10:06:36.601157Z","iopub.status.idle":"2022-07-21T10:06:36.629620Z","shell.execute_reply.started":"2022-07-21T10:06:36.601102Z","shell.execute_reply":"2022-07-21T10:06:36.628297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we model using Support Vector Machines which are supervised learning models with associated learning algorithms that analyze data used for classification and regression analysis. Given a set of training samples, each marked as belonging to one or the other of two categories, an SVM training algorithm builds a model that assigns new test samples to one category or the other, making it a non-probabilistic binary linear classifier. Reference Wikipedia.\n\nNote that the model generates a confidence score which is higher than Logistics Regression model.","metadata":{}},{"cell_type":"code","source":"# Support Vector Machines\n\nsvc = SVC()\nsvc.fit(X_train, Y_train)\nY_pred = svc.predict(X_test)\nacc_svc = round(svc.score(X_train, Y_train) * 100, 2)\nacc_svc","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.631418Z","iopub.execute_input":"2022-07-21T10:06:36.632218Z","iopub.status.idle":"2022-07-21T10:06:36.777379Z","shell.execute_reply.started":"2022-07-21T10:06:36.632168Z","shell.execute_reply":"2022-07-21T10:06:36.775912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In pattern recognition, the k-Nearest Neighbors algorithm (or k-NN for short) is a non-parametric method used for classification and regression. A sample is classified by a majority vote of its neighbors, with the sample being assigned to the class most common among its k nearest neighbors (k is a positive integer, typically small). If k = 1, then the object is simply assigned to the class of that single nearest neighbor. Reference Wikipedia.\n\nKNN confidence score is better than Logistics Regression but worse than SVM.","metadata":{}},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors = 3)\nknn.fit(X_train, Y_train)\nY_pred_knn = knn.predict(X_test)\nacc_knn = round(knn.score(X_train, Y_train) * 100, 2)\nacc_knn","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.780513Z","iopub.execute_input":"2022-07-21T10:06:36.781093Z","iopub.status.idle":"2022-07-21T10:06:36.850898Z","shell.execute_reply.started":"2022-07-21T10:06:36.781032Z","shell.execute_reply":"2022-07-21T10:06:36.849508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This model uses a decision tree as a predictive model which maps features (tree branches) to conclusions about the target value (tree leaves). Tree models where the target variable can take a finite set of values are called classification trees; in these tree structures, leaves represent class labels and branches represent conjunctions of features that lead to those class labels. Decision trees where the target variable can take continuous values (typically real numbers) are called regression trees. Reference Wikipedia.\n\nThe model confidence score is the highest among models evaluated so far.","metadata":{}},{"cell_type":"code","source":"# Decision Tree\n\ndecision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, Y_train)\nY_pred_dec = decision_tree.predict(X_test)\nacc_decision_tree = round(decision_tree.score(X_train, Y_train) * 100, 2)\nacc_decision_tree","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.853092Z","iopub.execute_input":"2022-07-21T10:06:36.853670Z","iopub.status.idle":"2022-07-21T10:06:36.875348Z","shell.execute_reply.started":"2022-07-21T10:06:36.853617Z","shell.execute_reply":"2022-07-21T10:06:36.874106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The next model Random Forests is one of the most popular. Random forests or random decision forests are an ensemble learning method for classification, regression and other tasks, that operate by constructing a multitude of decision trees (n_estimators=100) at training time and outputting the class that is the mode of the classes (classification) or mean prediction (regression) of the individual trees. Reference Wikipedia.\n\nThe model confidence score is the highest among models evaluated so far. We decide to use this model's output (Y_pred) for creating our competition submission of results.","metadata":{}},{"cell_type":"code","source":"# Random Forest\n\nrandom_forest = RandomForestClassifier(n_estimators=50)\nrandom_forest.fit(X_train, Y_train)\nY_pred_ran = random_forest.predict(X_test)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_train, Y_train) * 100, 2)\nacc_random_forest","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:36.877203Z","iopub.execute_input":"2022-07-21T10:06:36.877600Z","iopub.status.idle":"2022-07-21T10:06:37.072287Z","shell.execute_reply.started":"2022-07-21T10:06:36.877540Z","shell.execute_reply":"2022-07-21T10:06:37.071060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hypertuniing the model\n\nmodel = RandomForestClassifier(n_estimators= 80 ,max_depth=6 , max_features=8 ,min_samples_split=3 ,random_state=7)\nmodel.fit(X_train, Y_train)\npredictions = model.predict(X_test)\nmodel.score(X_train,Y_train)\nacc_model= round(model.score(X_train, Y_train) * 100, 2)\nacc_model","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:40:48.557416Z","iopub.execute_input":"2022-07-21T10:40:48.557872Z","iopub.status.idle":"2022-07-21T10:40:48.862008Z","shell.execute_reply.started":"2022-07-21T10:40:48.557835Z","shell.execute_reply":"2022-07-21T10:40:48.860649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED=42\nsingle_best_model = RandomForestClassifier(criterion='gini', \n                                           n_estimators=1100,\n                                           max_depth=6,\n                                           min_samples_split=4,\n                                           min_samples_leaf=5,\n                                           max_features=8,\n                                           oob_score=True,\n                                           random_state=SEED,\n                                           n_jobs=-1,\n                                           verbose=1)\nsingle_best_model.fit(X_train, Y_train)\nY_pred_random = single_best_model.predict(X_test)\nsingle_best_model.score(X_train, Y_train)\nacc_single_best_model = round(single_best_model.score(X_train, Y_train) * 100, 2)\nacc_single_best_model","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:45:34.632487Z","iopub.execute_input":"2022-07-21T10:45:34.632993Z","iopub.status.idle":"2022-07-21T10:45:40.927116Z","shell.execute_reply.started":"2022-07-21T10:45:34.632955Z","shell.execute_reply":"2022-07-21T10:45:40.925785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gradient Boosting Classifier\nfrom sklearn.ensemble import GradientBoostingClassifier\n\ngbk = GradientBoostingClassifier()\ngbk.fit(X_train, Y_train)\ny_pred = gbk.predict(X_train)\ny_pred_gbk = gbk.predict(X_test)\nacc_gbk = round(accuracy_score(y_pred, Y_train) * 100, 2)\nprint(acc_gbk)\ny_pred_gbk\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:53:37.245999Z","iopub.execute_input":"2022-07-21T10:53:37.246410Z","iopub.status.idle":"2022-07-21T10:53:37.442395Z","shell.execute_reply.started":"2022-07-21T10:53:37.246377Z","shell.execute_reply":"2022-07-21T10:53:37.441592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model evaluation\nWe can now rank our evaluation of all the models to choose the best one for our problem. While both Decision Tree and Random Forest score the same, we choose to use Random Forest as they correct for decision trees' habit of overfitting to their training set.","metadata":{}},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Logistic Regression', \n              'Random Forest', \n              'Decision Tree'],\n    'Score': [acc_svc, acc_knn, acc_log, \n              acc_random_forest, acc_decision_tree]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:06:37.074743Z","iopub.execute_input":"2022-07-21T10:06:37.075138Z","iopub.status.idle":"2022-07-21T10:06:37.089322Z","shell.execute_reply.started":"2022-07-21T10:06:37.075104Z","shell.execute_reply":"2022-07-21T10:06:37.088447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        \"PassengerId\": test[\"PassengerId\"],\n        \"Survived\": Y_pred_random\n    })\nsubmission.to_csv('titanic_submission_9.csv',index = False, header=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:53:56.525149Z","iopub.execute_input":"2022-07-21T10:53:56.525589Z","iopub.status.idle":"2022-07-21T10:53:56.534814Z","shell.execute_reply.started":"2022-07-21T10:53:56.525541Z","shell.execute_reply":"2022-07-21T10:53:56.533467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"References\nThis notebook has been created based on great work done solving the Titanic competition and other sources.\n\nTitanic Data Science Solutions(https://www.kaggle.com/code/startupsci/titanic-data-science-solutions)\n\nFor Null values of Age & Random Forest Tunning (https://www.kaggle.com/code/odaymourad/learn-overfitting-and-underfitting-79-4-score) ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}