{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing important libraries\nimport numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.446976Z","iopub.execute_input":"2022-07-30T06:20:30.447389Z","iopub.status.idle":"2022-07-30T06:20:30.453727Z","shell.execute_reply.started":"2022-07-30T06:20:30.447355Z","shell.execute_reply":"2022-07-30T06:20:30.452407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train dataset\ntrain=pd.read_csv('../input/titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.459147Z","iopub.execute_input":"2022-07-30T06:20:30.459526Z","iopub.status.idle":"2022-07-30T06:20:30.473425Z","shell.execute_reply.started":"2022-07-30T06:20:30.459491Z","shell.execute_reply":"2022-07-30T06:20:30.472394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test dataset\ntest=pd.read_csv('../input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.478863Z","iopub.execute_input":"2022-07-30T06:20:30.479678Z","iopub.status.idle":"2022-07-30T06:20:30.489409Z","shell.execute_reply.started":"2022-07-30T06:20:30.479640Z","shell.execute_reply":"2022-07-30T06:20:30.488119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combining the train and test into a new dataframe\ndf=pd.concat([train,test],ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.491588Z","iopub.execute_input":"2022-07-30T06:20:30.492681Z","iopub.status.idle":"2022-07-30T06:20:30.501922Z","shell.execute_reply.started":"2022-07-30T06:20:30.492627Z","shell.execute_reply":"2022-07-30T06:20:30.501078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First five rows of the combined data\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.503857Z","iopub.execute_input":"2022-07-30T06:20:30.504847Z","iopub.status.idle":"2022-07-30T06:20:30.524597Z","shell.execute_reply.started":"2022-07-30T06:20:30.504803Z","shell.execute_reply":"2022-07-30T06:20:30.522886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape of the data\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.526915Z","iopub.execute_input":"2022-07-30T06:20:30.527781Z","iopub.status.idle":"2022-07-30T06:20:30.537428Z","shell.execute_reply.started":"2022-07-30T06:20:30.527730Z","shell.execute_reply":"2022-07-30T06:20:30.536385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Info of the data\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.541918Z","iopub.execute_input":"2022-07-30T06:20:30.542282Z","iopub.status.idle":"2022-07-30T06:20:30.562599Z","shell.execute_reply.started":"2022-07-30T06:20:30.542253Z","shell.execute_reply":"2022-07-30T06:20:30.561449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Summary Statistics\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.563954Z","iopub.execute_input":"2022-07-30T06:20:30.564275Z","iopub.status.idle":"2022-07-30T06:20:30.597632Z","shell.execute_reply.started":"2022-07-30T06:20:30.564246Z","shell.execute_reply":"2022-07-30T06:20:30.596825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Corelation matrix\ndf.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.599109Z","iopub.execute_input":"2022-07-30T06:20:30.599426Z","iopub.status.idle":"2022-07-30T06:20:30.617473Z","shell.execute_reply.started":"2022-07-30T06:20:30.599396Z","shell.execute_reply":"2022-07-30T06:20:30.616215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Heatmap for the correlation matrix\nsns.heatmap(df.corr(),cmap='YlGnBu')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.619112Z","iopub.execute_input":"2022-07-30T06:20:30.619448Z","iopub.status.idle":"2022-07-30T06:20:30.916397Z","shell.execute_reply.started":"2022-07-30T06:20:30.619417Z","shell.execute_reply":"2022-07-30T06:20:30.915327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for missing values\ndf.isnull().sum().sort_values()\n# Data has missing values","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.922055Z","iopub.execute_input":"2022-07-30T06:20:30.922428Z","iopub.status.idle":"2022-07-30T06:20:30.936139Z","shell.execute_reply.started":"2022-07-30T06:20:30.922396Z","shell.execute_reply":"2022-07-30T06:20:30.934888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Univariate Analysis","metadata":{}},{"cell_type":"code","source":"# We would first extract all the numerical columns\ndf.select_dtypes(np.number).drop(columns=['PassengerId','Survived','Pclass']).columns","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.938073Z","iopub.execute_input":"2022-07-30T06:20:30.938570Z","iopub.status.idle":"2022-07-30T06:20:30.951062Z","shell.execute_reply.started":"2022-07-30T06:20:30.938504Z","shell.execute_reply":"2022-07-30T06:20:30.949638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nums=['Age', 'SibSp', 'Parch', 'Fare'] # All the numerical columns","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.952302Z","iopub.execute_input":"2022-07-30T06:20:30.952682Z","iopub.status.idle":"2022-07-30T06:20:30.957681Z","shell.execute_reply.started":"2022-07-30T06:20:30.952647Z","shell.execute_reply":"2022-07-30T06:20:30.956596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making multiple distribution plots for Univariate analysis of the numerical columns\nrows=2\ncols=2\ncounter=1\nplt.rcParams['figure.figsize']=[10,8]\nfor i in nums:\n    plt.subplot(rows,cols,counter)\n    sns.distplot(df[i])\n    counter+=1\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:30.959245Z","iopub.execute_input":"2022-07-30T06:20:30.960488Z","iopub.status.idle":"2022-07-30T06:20:32.075233Z","shell.execute_reply.started":"2022-07-30T06:20:30.960438Z","shell.execute_reply":"2022-07-30T06:20:32.074149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we see the categorical variables\ncats=['Survived','Pclass','Sex','Embarked']","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.076744Z","iopub.execute_input":"2022-07-30T06:20:32.077089Z","iopub.status.idle":"2022-07-30T06:20:32.082019Z","shell.execute_reply.started":"2022-07-30T06:20:32.077057Z","shell.execute_reply":"2022-07-30T06:20:32.081034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we will make multiple countplots for univariate analysis of the categorical variable\nrows=2\ncols=2\ncounter=1\nplt.rcParams['figure.figsize']=[10,8]\nfor i in cats:\n    plt.subplot(rows,cols,counter)\n    sns.countplot(df[i])\n    counter+=1\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.083299Z","iopub.execute_input":"2022-07-30T06:20:32.084101Z","iopub.status.idle":"2022-07-30T06:20:32.568730Z","shell.execute_reply.started":"2022-07-30T06:20:32.084065Z","shell.execute_reply":"2022-07-30T06:20:32.567301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Insights from the Univariate Analysis\n\n* The Age column seems to be the nearest to a normal distribution\n* Most of the people embarked were in the rangeb of 20 and 40\n* Most poeple were travelinng alone or were travelling with 1 person\n* Most people travelling didn't have any children with them or had 1 child with them\n* The fare was mostly between 0 and 100 pounds\n* People with 0 fare could be crew members\n* A lot less people survived than people who died\n* There were more passengers with in 3rd passenger class than any other passenger class\n* There were more males than females on the ship. This could indicate that more males died on the ship than females\n* Most people on the ship embarked from Southampton. This could mean that most of the people who died were from Southampton","metadata":{}},{"cell_type":"code","source":"# Before procedding on with our bivariate analysis\n# We would do some feature engineering\n# Handle our missing values\n# Drop some redundant columns","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.570217Z","iopub.execute_input":"2022-07-30T06:20:32.570549Z","iopub.status.idle":"2022-07-30T06:20:32.576189Z","shell.execute_reply.started":"2022-07-30T06:20:32.570507Z","shell.execute_reply":"2022-07-30T06:20:32.574628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering, and Handling of Missing values","metadata":{}},{"cell_type":"code","source":"# The first feature that we would engineer are the Sibsp and parch column\n# What we would do is that we will add these two columns and add 1 to it to take into account the number of people travelling together\ndf['FamilyMembers']=df.SibSp+df.Parch+1\n# We have now created our new feature","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.578105Z","iopub.execute_input":"2022-07-30T06:20:32.578712Z","iopub.status.idle":"2022-07-30T06:20:32.592270Z","shell.execute_reply.started":"2022-07-30T06:20:32.578667Z","shell.execute_reply":"2022-07-30T06:20:32.591032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Since FamilyMembers column is serving our purpose\n# we will drop the SibSp and Parch columns\ndf.drop(columns=['SibSp','Parch'],inplace=True)# Both the columns are now dropped","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.594056Z","iopub.execute_input":"2022-07-30T06:20:32.594893Z","iopub.status.idle":"2022-07-30T06:20:32.603500Z","shell.execute_reply.started":"2022-07-30T06:20:32.594847Z","shell.execute_reply":"2022-07-30T06:20:32.602394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can see that there are some categorical columns that are of integer type\n# We would change them to object\ndf.Pclass=df.Pclass.astype('object')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.604699Z","iopub.execute_input":"2022-07-30T06:20:32.605417Z","iopub.status.idle":"2022-07-30T06:20:32.616612Z","shell.execute_reply.started":"2022-07-30T06:20:32.605376Z","shell.execute_reply":"2022-07-30T06:20:32.615725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We would be dropping the passenger id and ticket column because we feel they are redundant\ndf.drop(columns=['PassengerId','Ticket'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.617868Z","iopub.execute_input":"2022-07-30T06:20:32.618356Z","iopub.status.idle":"2022-07-30T06:20:32.630495Z","shell.execute_reply.started":"2022-07-30T06:20:32.618325Z","shell.execute_reply":"2022-07-30T06:20:32.629424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# In terms of missing values\n# We will first handle the cabin column\n# In this we would also bin the column in two categories:- 'Cabin alloted' and 'Cabin not alloted'\n# For this we would replace the non null entries with 'Cabin alloted'\n# And we will fill the null entries with 'Cabin Not-alloted'\ndf.loc[df.Cabin.isnull()==False,'Cabin']='Cabin Alloted'\ndf.Cabin.fillna('Cabin not alloted',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.635977Z","iopub.execute_input":"2022-07-30T06:20:32.636564Z","iopub.status.idle":"2022-07-30T06:20:32.644004Z","shell.execute_reply.started":"2022-07-30T06:20:32.636507Z","shell.execute_reply":"2022-07-30T06:20:32.642870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()\n# We can see that there are 1 and 2 null values in Fare and Embarked respectively\n# We would use mean imputation and mode imputation to fill them up","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.645133Z","iopub.execute_input":"2022-07-30T06:20:32.645955Z","iopub.status.idle":"2022-07-30T06:20:32.658465Z","shell.execute_reply.started":"2022-07-30T06:20:32.645918Z","shell.execute_reply":"2022-07-30T06:20:32.657669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Fare.fillna(df.Fare.mean(),inplace=True)\ndf.Embarked.fillna(df.Embarked.mode().values[0],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.659682Z","iopub.execute_input":"2022-07-30T06:20:32.660167Z","iopub.status.idle":"2022-07-30T06:20:32.671377Z","shell.execute_reply.started":"2022-07-30T06:20:32.660139Z","shell.execute_reply":"2022-07-30T06:20:32.670334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()\n# Now only the age column has null values\n# Before filling that we would be doing feature engineering to extract the title from the names of the people(this is why we didn't drop the colum yet)\n# Then we would fill the null values in age according to the Pclass and title","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.672742Z","iopub.execute_input":"2022-07-30T06:20:32.673070Z","iopub.status.idle":"2022-07-30T06:20:32.688588Z","shell.execute_reply.started":"2022-07-30T06:20:32.673040Z","shell.execute_reply":"2022-07-30T06:20:32.687403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting the name columns in two halves\ndf[['delete','work']]=df.Name.str.split(', ',expand=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.689993Z","iopub.execute_input":"2022-07-30T06:20:32.691014Z","iopub.status.idle":"2022-07-30T06:20:32.701058Z","shell.execute_reply.started":"2022-07-30T06:20:32.690977Z","shell.execute_reply":"2022-07-30T06:20:32.699828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Since we have the column where we can easily extract the title we would drop the 'delete' column and the 'Name' column\ndf.drop(columns=['Name','delete'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.702610Z","iopub.execute_input":"2022-07-30T06:20:32.703660Z","iopub.status.idle":"2022-07-30T06:20:32.715186Z","shell.execute_reply.started":"2022-07-30T06:20:32.703594Z","shell.execute_reply":"2022-07-30T06:20:32.714094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We would now separate titles from the name\nlst=[]\nfor x in df.work:\n    lst.append(x.split('. ')[0])\n# Now since our list of title is created we would create a new feature 'Title' and put the list that we got as the column\ndf['Title']=lst #Title column is now created","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.716385Z","iopub.execute_input":"2022-07-30T06:20:32.717183Z","iopub.status.idle":"2022-07-30T06:20:32.732593Z","shell.execute_reply.started":"2022-07-30T06:20:32.717147Z","shell.execute_reply":"2022-07-30T06:20:32.731649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping extra columns that we don't need\ndf.drop(columns=['work'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.734483Z","iopub.execute_input":"2022-07-30T06:20:32.735460Z","iopub.status.idle":"2022-07-30T06:20:32.746265Z","shell.execute_reply.started":"2022-07-30T06:20:32.735422Z","shell.execute_reply":"2022-07-30T06:20:32.744834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we would proceed to filling null values in age column\n# First we will see the unique values in Title column\ndf.Title.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.748030Z","iopub.execute_input":"2022-07-30T06:20:32.748602Z","iopub.status.idle":"2022-07-30T06:20:32.759780Z","shell.execute_reply.started":"2022-07-30T06:20:32.748472Z","shell.execute_reply":"2022-07-30T06:20:32.758880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we would create a list where we would keep all those titles where we would like bin them as others\n# We would keep the primary titles like 'Mr','Mrs','Master',etc. intact while all the other title would fall under a bin 'Others'\nothers=[ 'Don', 'Rev', 'Dr', 'Mme',\n       'Major', 'Lady', 'Sir', 'Mlle', 'Col', 'Capt', 'the Countess',\n       'Jonkheer', 'Dona']","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.760899Z","iopub.execute_input":"2022-07-30T06:20:32.761614Z","iopub.status.idle":"2022-07-30T06:20:32.770399Z","shell.execute_reply.started":"2022-07-30T06:20:32.761573Z","shell.execute_reply":"2022-07-30T06:20:32.769546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we would create a function to place the titles under the bins according to the list we created above\ndef title_bin(x):\n    if x in others:\n        return 'Others'\n    else:\n        return x\ndf.Title=df.Title.apply(title_bin) # Binning the title column","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.771792Z","iopub.execute_input":"2022-07-30T06:20:32.772325Z","iopub.status.idle":"2022-07-30T06:20:32.782269Z","shell.execute_reply.started":"2022-07-30T06:20:32.772290Z","shell.execute_reply":"2022-07-30T06:20:32.781203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We would fill the age column by groupwise median imputation according to title and pclass\n# First we would find the median age for each title in each class\npd.DataFrame(df.groupby(['Title','Pclass'])['Age'].median())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.783685Z","iopub.execute_input":"2022-07-30T06:20:32.784222Z","iopub.status.idle":"2022-07-30T06:20:32.804089Z","shell.execute_reply.started":"2022-07-30T06:20:32.784180Z","shell.execute_reply":"2022-07-30T06:20:32.802949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we would do the groupwise median imputation\ndf.loc[((df[\"Title\"]==\"Master\") & (df[\"Pclass\"]==1)) & (df[\"Age\"].isnull()),\"Age\"]=6.0\ndf.loc[((df[\"Title\"]==\"Master\") & (df[\"Pclass\"]==2)) & (df[\"Age\"].isnull()),\"Age\"]=2.0\ndf.loc[((df[\"Title\"]==\"Master\") & (df[\"Pclass\"]==3)) & (df[\"Age\"].isnull()),\"Age\"]=6.0\ndf.loc[((df[\"Title\"]==\"Miss\") & (df[\"Pclass\"]==1)) & (df[\"Age\"].isnull()),\"Age\"]=30.0\ndf.loc[((df[\"Title\"]==\"Miss\") & (df[\"Pclass\"]==2)) & (df[\"Age\"].isnull()),\"Age\"]=20.0\ndf.loc[((df[\"Title\"]==\"Miss\") & (df[\"Pclass\"]==3)) & (df[\"Age\"].isnull()),\"Age\"]=18.0\ndf.loc[((df[\"Title\"]==\"Mr\") & (df[\"Pclass\"]==1)) & (df[\"Age\"].isnull()),\"Age\"]=41.5\ndf.loc[((df[\"Title\"]==\"Mr\") & (df[\"Pclass\"]==2)) & (df[\"Age\"].isnull()),\"Age\"]=30.0\ndf.loc[((df[\"Title\"]==\"Mr\") & (df[\"Pclass\"]==3)) & (df[\"Age\"].isnull()),\"Age\"]=26.0\ndf.loc[((df[\"Title\"]==\"Mrs\") & (df[\"Pclass\"]==1)) & (df[\"Age\"].isnull()),\"Age\"]=45.0\ndf.loc[((df[\"Title\"]==\"Mrs\") & (df[\"Pclass\"]==2)) & (df[\"Age\"].isnull()),\"Age\"]=30.5\ndf.loc[((df[\"Title\"]==\"Mrs\") & (df[\"Pclass\"]==3)) & (df[\"Age\"].isnull()),\"Age\"]=31.0\ndf.loc[(df[\"Title\"]==\"Ms\") &  (df[\"Age\"].isnull()),\"Age\"]=28.0\ndf.loc[((df[\"Title\"]==\"Others\") & (df[\"Pclass\"]==1)) & (df[\"Age\"].isnull()),\"Age\"]=47.0\ndf.loc[((df[\"Title\"]==\"Others\") & (df[\"Pclass\"]==2)) & (df[\"Age\"].isnull()),\"Age\"]=41.5","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.805929Z","iopub.execute_input":"2022-07-30T06:20:32.806651Z","iopub.status.idle":"2022-07-30T06:20:32.849882Z","shell.execute_reply.started":"2022-07-30T06:20:32.806602Z","shell.execute_reply":"2022-07-30T06:20:32.848886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum() # Age column is also now filled","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.851340Z","iopub.execute_input":"2022-07-30T06:20:32.851649Z","iopub.status.idle":"2022-07-30T06:20:32.861000Z","shell.execute_reply.started":"2022-07-30T06:20:32.851621Z","shell.execute_reply":"2022-07-30T06:20:32.860179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Since our Data is now cleaned we can now proceed with our bivariate analysis","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.862316Z","iopub.execute_input":"2022-07-30T06:20:32.862911Z","iopub.status.idle":"2022-07-30T06:20:32.871881Z","shell.execute_reply.started":"2022-07-30T06:20:32.862808Z","shell.execute_reply":"2022-07-30T06:20:32.870921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bivariate Analysis","metadata":{}},{"cell_type":"code","source":"# First we will start with the numerical features\ndf.select_dtypes(np.number).drop(columns='Survived').columns","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.873089Z","iopub.execute_input":"2022-07-30T06:20:32.873423Z","iopub.status.idle":"2022-07-30T06:20:32.887211Z","shell.execute_reply.started":"2022-07-30T06:20:32.873393Z","shell.execute_reply":"2022-07-30T06:20:32.886327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nums=['Age', 'Fare', 'FamilyMembers'] # Numerical features","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.888954Z","iopub.execute_input":"2022-07-30T06:20:32.889695Z","iopub.status.idle":"2022-07-30T06:20:32.897749Z","shell.execute_reply.started":"2022-07-30T06:20:32.889649Z","shell.execute_reply":"2022-07-30T06:20:32.896692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making multiple boxplots\nrows=2\ncols=2\ncounter=1\nplt.rcParams['figure.figsize']=[10,8]\nfor i in nums:\n    plt.subplot(rows,cols,counter)\n    sns.boxplot(x='Survived',y=i,data=df)\n    counter+=1\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:32.899063Z","iopub.execute_input":"2022-07-30T06:20:32.899952Z","iopub.status.idle":"2022-07-30T06:20:33.361521Z","shell.execute_reply.started":"2022-07-30T06:20:32.899921Z","shell.execute_reply":"2022-07-30T06:20:33.360429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we will take the Categorical Features\ndf.select_dtypes(object).columns","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:33.363047Z","iopub.execute_input":"2022-07-30T06:20:33.363491Z","iopub.status.idle":"2022-07-30T06:20:33.373106Z","shell.execute_reply.started":"2022-07-30T06:20:33.363446Z","shell.execute_reply":"2022-07-30T06:20:33.371848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat=['Pclass', 'Sex', 'Cabin', 'Embarked', 'Title'] # All the categorical column","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:33.374801Z","iopub.execute_input":"2022-07-30T06:20:33.375241Z","iopub.status.idle":"2022-07-30T06:20:33.383808Z","shell.execute_reply.started":"2022-07-30T06:20:33.375199Z","shell.execute_reply":"2022-07-30T06:20:33.382885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting multiple crosstabs as barplots\nplt.rcParams['figure.figsize']=[5,5]\nfor i in cat:\n    pd.crosstab(df[i],df.Survived).plot(kind='bar')\n    plt.xticks(rotation=0)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:33.385127Z","iopub.execute_input":"2022-07-30T06:20:33.385558Z","iopub.status.idle":"2022-07-30T06:20:34.387651Z","shell.execute_reply.started":"2022-07-30T06:20:33.385502Z","shell.execute_reply":"2022-07-30T06:20:34.386286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Insights from the Bivariate Analysis\n\n* Younger people had a higher chance of survival than older people\n* People who paid higher fare had a considerable higher chance of survival\n* People with higher number of family members travelling together had a considerably lower chance of survival\n* People travelling in the first class seem to have a higher chance of surviving the disaster\n* A ratio of males dying is a lot more than females not surviving the disaster\n* A consider number of people died who were not alloted cabins than the people who were alloted cabins\n* People dying from Southampton is considerbaly higher than people from Cherbough and Queenstown\n* From the title barplot we can clearly see that the number of males dying was considerably higher than women and children","metadata":{}},{"cell_type":"markdown","source":"# Encoding, Splitting and Scaling of the data","metadata":{}},{"cell_type":"code","source":"# We would use the one hot encoding method and create dummy variables for the categorical columns\ndummy=pd.get_dummies(df,drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.389088Z","iopub.execute_input":"2022-07-30T06:20:34.389661Z","iopub.status.idle":"2022-07-30T06:20:34.403789Z","shell.execute_reply.started":"2022-07-30T06:20:34.389626Z","shell.execute_reply":"2022-07-30T06:20:34.402263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we will split the dummy dataframe into train and test\n# Since already knew the size of train and test data\n# we would use the iloc method to split the same\ndf_train=dummy.iloc[:891,:] # Train dummy data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.405433Z","iopub.execute_input":"2022-07-30T06:20:34.406372Z","iopub.status.idle":"2022-07-30T06:20:34.410925Z","shell.execute_reply.started":"2022-07-30T06:20:34.406335Z","shell.execute_reply":"2022-07-30T06:20:34.409729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test=dummy.iloc[891:,:] # Test dummy data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.412908Z","iopub.execute_input":"2022-07-30T06:20:34.413732Z","iopub.status.idle":"2022-07-30T06:20:34.424373Z","shell.execute_reply.started":"2022-07-30T06:20:34.413685Z","shell.execute_reply":"2022-07-30T06:20:34.423463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.drop(columns='Survived',inplace=True) # Dropping the target variable from the test data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.426020Z","iopub.execute_input":"2022-07-30T06:20:34.426737Z","iopub.status.idle":"2022-07-30T06:20:34.436929Z","shell.execute_reply.started":"2022-07-30T06:20:34.426691Z","shell.execute_reply":"2022-07-30T06:20:34.435819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we will scale the data\nfrom sklearn.preprocessing import MinMaxScaler\nsc=MinMaxScaler()\ndf_train.iloc[:,1:4]=sc.fit_transform(df_train.iloc[:,1:4])\ndf_test.iloc[:,0:3]=sc.transform(df_test.iloc[:,0:3])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.438113Z","iopub.execute_input":"2022-07-30T06:20:34.439067Z","iopub.status.idle":"2022-07-30T06:20:34.454164Z","shell.execute_reply.started":"2022-07-30T06:20:34.439027Z","shell.execute_reply":"2022-07-30T06:20:34.453058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now initializing our X and Y variables in train and test data\nx_train=df_train.drop(columns='Survived')\ny_train=df_train.Survived\nx_test=df_test","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.457635Z","iopub.execute_input":"2022-07-30T06:20:34.458137Z","iopub.status.idle":"2022-07-30T06:20:34.464073Z","shell.execute_reply.started":"2022-07-30T06:20:34.458107Z","shell.execute_reply":"2022-07-30T06:20:34.463227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Bulding","metadata":{}},{"cell_type":"markdown","source":"## Logistic Regression","metadata":{}},{"cell_type":"code","source":"# We will first use Logistic Regression\nfrom sklearn.linear_model import LogisticRegression\nlg=LogisticRegression()\nmodel=lg.fit(x_train,y_train)\ny_test=model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.465365Z","iopub.execute_input":"2022-07-30T06:20:34.465885Z","iopub.status.idle":"2022-07-30T06:20:34.509073Z","shell.execute_reply.started":"2022-07-30T06:20:34.465846Z","shell.execute_reply":"2022-07-30T06:20:34.507726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction=pd.DataFrame(y_test,test.PassengerId).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.510805Z","iopub.execute_input":"2022-07-30T06:20:34.511993Z","iopub.status.idle":"2022-07-30T06:20:34.521126Z","shell.execute_reply.started":"2022-07-30T06:20:34.511934Z","shell.execute_reply":"2022-07-30T06:20:34.519627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Decision Tree Classifier","metadata":{}},{"cell_type":"code","source":"# Decision Tree Classifier\nfrom sklearn.tree import DecisionTreeClassifier\ndt=DecisionTreeClassifier()\nmodel=dt.fit(x_train,y_train)\ny_test=model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.523754Z","iopub.execute_input":"2022-07-30T06:20:34.524817Z","iopub.status.idle":"2022-07-30T06:20:34.544522Z","shell.execute_reply.started":"2022-07-30T06:20:34.524758Z","shell.execute_reply":"2022-07-30T06:20:34.543232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction=pd.DataFrame(y_test,test.PassengerId).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.546493Z","iopub.execute_input":"2022-07-30T06:20:34.547297Z","iopub.status.idle":"2022-07-30T06:20:34.556516Z","shell.execute_reply.started":"2022-07-30T06:20:34.547245Z","shell.execute_reply":"2022-07-30T06:20:34.555023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forrest","metadata":{}},{"cell_type":"code","source":"# Random Forrest\nfrom sklearn.ensemble import RandomForestClassifier\nrf=RandomForestClassifier()\nmodel=rf.fit(x_train,y_train)\ny_test=model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.559317Z","iopub.execute_input":"2022-07-30T06:20:34.560389Z","iopub.status.idle":"2022-07-30T06:20:34.839541Z","shell.execute_reply.started":"2022-07-30T06:20:34.560326Z","shell.execute_reply":"2022-07-30T06:20:34.838316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction=pd.DataFrame(y_test,test.PassengerId).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.840959Z","iopub.execute_input":"2022-07-30T06:20:34.841320Z","iopub.status.idle":"2022-07-30T06:20:34.848786Z","shell.execute_reply.started":"2022-07-30T06:20:34.841289Z","shell.execute_reply":"2022-07-30T06:20:34.847441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Light GBM","metadata":{}},{"cell_type":"code","source":"# LightGBM\nfrom lightgbm import LGBMClassifier\nlgb=LGBMClassifier()\nmodel=lgb.fit(x_train,y_train)\ny_test=model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.857407Z","iopub.execute_input":"2022-07-30T06:20:34.857836Z","iopub.status.idle":"2022-07-30T06:20:34.950874Z","shell.execute_reply.started":"2022-07-30T06:20:34.857801Z","shell.execute_reply":"2022-07-30T06:20:34.949727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction=pd.DataFrame(y_test,test.PassengerId).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:20:34.955815Z","iopub.execute_input":"2022-07-30T06:20:34.956732Z","iopub.status.idle":"2022-07-30T06:20:34.967742Z","shell.execute_reply.started":"2022-07-30T06:20:34.956690Z","shell.execute_reply":"2022-07-30T06:20:34.966483Z"},"trusted":true},"execution_count":null,"outputs":[]}]}