{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello, everybody! This is my first attempt at a kaggle competition so any suggestions or advices are much appreciated. \n\nThe project is structured in three different notebooks: \n- Univariate Analysis \n- Multivariate Analysis \n- Preprocessing and Modeling \n\nI hope you find it interesting! ","metadata":{}},{"cell_type":"markdown","source":"# Titanic - Univariate Analysis","metadata":{}},{"cell_type":"markdown","source":"<a id=\"import\"></a>\n### Importing Libraries","metadata":{"tags":[]}},{"cell_type":"code","source":"# Data management \nimport pandas as pd \nimport numpy as np \n\n# Statistics and ML\nimport statsmodels.api as sm\nfrom scipy import stats\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import MissingIndicator\n\n# Data visualization \nimport matplotlib.pyplot as plt\nfrom matplotlib.lines import Line2D\nimport seaborn as sns \n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:11.867901Z","iopub.execute_input":"2022-08-05T09:23:11.868758Z","iopub.status.idle":"2022-08-05T09:23:11.875441Z","shell.execute_reply.started":"2022-08-05T09:23:11.868715Z","shell.execute_reply":"2022-08-05T09:23:11.874458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"loading\"></a>\n### Loading data","metadata":{"tags":[]}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/titanic/train.csv\") \nprint(\" \") \nprint(pd.Series({\"Memory usage\": \"{:.2f} MB\".format(train.memory_usage().sum()/(1024*1024)),\n                 \"Dataset shape\": \"{}\".format(train.shape)}).to_string(),\n                 end = \"\\n\\n\") \nprint(train.dtypes, end = \"\\n\\n\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:11.890837Z","iopub.execute_input":"2022-08-05T09:23:11.892006Z","iopub.status.idle":"2022-08-05T09:23:11.951618Z","shell.execute_reply.started":"2022-08-05T09:23:11.891952Z","shell.execute_reply":"2022-08-05T09:23:11.950385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"../input/titanic/test.csv\") \nprint(\" \") \nprint(pd.Series({\"Memory usage\": \"{:.2f} MB\".format(test.memory_usage().sum()/(1024*1024)),\n                 \"Dataset shape\": \"{}\".format(test.shape)}).to_string(),\n                 end = \"\\n\\n\")  \nprint(test.dtypes, end = \"\\n\\n\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:11.954020Z","iopub.execute_input":"2022-08-05T09:23:11.954802Z","iopub.status.idle":"2022-08-05T09:23:11.991088Z","shell.execute_reply.started":"2022-08-05T09:23:11.954753Z","shell.execute_reply":"2022-08-05T09:23:11.989872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"indexing\"></a>\n### Setting index","metadata":{"tags":[]}},{"cell_type":"code","source":"train = train.set_index('PassengerId', drop = True)\ntest = test.set_index('PassengerId', drop = True)\n\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:11.993306Z","iopub.execute_input":"2022-08-05T09:23:11.994072Z","iopub.status.idle":"2022-08-05T09:23:12.018081Z","shell.execute_reply.started":"2022-08-05T09:23:11.994024Z","shell.execute_reply":"2022-08-05T09:23:12.017205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"univ_sec\"></a>\n## Univariate Analysis","metadata":{}},{"cell_type":"markdown","source":"<a id=\"missing\"></a>\n### Missing Values","metadata":{"tags":[]}},{"cell_type":"markdown","source":"#### Training set","metadata":{}},{"cell_type":"code","source":"print(train.isnull().sum())\nsns.heatmap(train.isnull())","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:12.020295Z","iopub.execute_input":"2022-08-05T09:23:12.021780Z","iopub.status.idle":"2022-08-05T09:23:12.467987Z","shell.execute_reply.started":"2022-08-05T09:23:12.021730Z","shell.execute_reply":"2022-08-05T09:23:12.466532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Percentage of missing values in the training set')\nprint((train.isnull().sum() / train.shape[0]) * 100)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:12.469726Z","iopub.execute_input":"2022-08-05T09:23:12.470089Z","iopub.status.idle":"2022-08-05T09:23:12.484035Z","shell.execute_reply.started":"2022-08-05T09:23:12.470059Z","shell.execute_reply":"2022-08-05T09:23:12.482352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Test set","metadata":{}},{"cell_type":"code","source":"print(test.isnull().sum())\nsns.heatmap(test.isnull())","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:12.485549Z","iopub.execute_input":"2022-08-05T09:23:12.486003Z","iopub.status.idle":"2022-08-05T09:23:12.905744Z","shell.execute_reply.started":"2022-08-05T09:23:12.485969Z","shell.execute_reply":"2022-08-05T09:23:12.903670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Percentage of missing values in the test set')\nprint((test.isnull().sum() / test.shape[0]) * 100)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:12.907137Z","iopub.execute_input":"2022-08-05T09:23:12.907518Z","iopub.status.idle":"2022-08-05T09:23:12.918522Z","shell.execute_reply.started":"2022-08-05T09:23:12.907488Z","shell.execute_reply":"2022-08-05T09:23:12.917186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Very similar missing patterns in both sets. \n- Cabin is probably not a useful predictor perse due to the elevated number of missing values. \n- It seems that the age missing pattern is evenly distributed across the cases. ","metadata":{}},{"cell_type":"markdown","source":"<a id=\"target\"></a>\n### Target - Survive","metadata":{"tags":[]}},{"cell_type":"code","source":"surv_tab = pd.DataFrame({'Count': train.Survived.value_counts(), 'Percentage': train.Survived.value_counts(normalize = True).mul(100)})\nprint(surv_tab) \ng1 = sns.countplot(x = train.Survived, palette = 'Set1')\ng1.set_title('Number of survivals') \n\ng1.set(xticklabels=[\"No\", \"Yes\"])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:12.919873Z","iopub.execute_input":"2022-08-05T09:23:12.920397Z","iopub.status.idle":"2022-08-05T09:23:13.134072Z","shell.execute_reply.started":"2022-08-05T09:23:12.920348Z","shell.execute_reply":"2022-08-05T09:23:13.132627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"age\"></a>\n### Age","metadata":{"tags":[]}},{"cell_type":"markdown","source":"#### Exploration ","metadata":{}},{"cell_type":"code","source":"print('\\nTraining set:')\nprint(train.Age.describe())\nprint(f\"\\nSkewness: {stats.skew(train.Age, nan_policy = 'omit')}\\nKurtosis: {stats.kurtosis(train.Age, nan_policy = 'omit')}\")\n\nprint('\\nTest set:')\nprint(test.Age.describe())\nprint(f\"\\nSkewness: {stats.skew(test.Age, nan_policy = 'omit')}\\nKurtosis: {stats.kurtosis(test.Age, nan_policy = 'omit')}\")\n\n\n# Histogram \nfig, axs = plt.subplots(1, 2, figsize = (14,6))\ng1 = sns.histplot(train.Age, ax = axs[0]) \ng1 = sns.histplot(test.Age, ax = axs[0], color = 'yellow') \ng1.set_title(\"Age in Training set and Test Set\")\n\n# Standardizing\n# *We will not insert the scaled variables until the preprocessing stage but we need them for the qq-plot\nscaler = StandardScaler() \nz_Age_train = scaler.fit_transform(train[['Age']])\nz_Age_test = scaler.transform(test[['Age']]) # Using the same scaler\n\n\n# Q-Q plot \ng2 = sm.qqplot(z_Age_train[~np.isnan(z_Age_train)], line='45', ax = axs[1])\ng2 = sm.qqplot(z_Age_test[~np.isnan(z_Age_test)], line='45', markerfacecolor = 'yellow', markeredgecolor='yellow', alpha = 0.5, ax = axs[1])\ng2.suptitle('Q-Q plot Age training set/test set', x = 0.77, y = 0.93)\n\n\nlegend = [Line2D([0], [0], marker='s', color='w', label='Test set', markerfacecolor='yellow', markersize=5),\n          Line2D([0], [0], marker='s', color='w', label='Training set', markerfacecolor='royalblue', markersize=5)]\nfig.legend(handles=legend, fontsize = 'large', bbox_to_anchor=(0.15, 0))\n\n\nfig.tight_layout() \n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:13.137452Z","iopub.execute_input":"2022-08-05T09:23:13.137877Z","iopub.status.idle":"2022-08-05T09:23:13.771062Z","shell.execute_reply.started":"2022-08-05T09:23:13.137842Z","shell.execute_reply":"2022-08-05T09:23:13.769730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- We can assume normal distribution\n- There are not big differences between the train set and the test set \n- The majority of passegers are between 20 and 40 years old (around 50%) and the mean is 30. ","metadata":{}},{"cell_type":"markdown","source":"<a id=\"fare\"></a>\n### Fare","metadata":{"tags":[]}},{"cell_type":"markdown","source":"#### Exploration","metadata":{}},{"cell_type":"code","source":"print('\\nTraining set:')\nprint(train.Fare.describe())\nprint(f\"\\nSkewness: {stats.skew(train.Fare)}\\nKurtosis: {stats.kurtosis(train.Fare)}\")\n\nprint('\\nTest set:')\nprint(test.Fare.describe())\nprint(f\"\\nSkewness: {stats.skew(test.Fare)}\\nKurtosis: {stats.kurtosis(test.Fare)}\")\n\n\n# Histogram\nfig, axs = plt.subplots(1, 2, figsize = (14,6))\n\ng1 = sns.histplot(train.Fare, ax = axs[0])\ng1 = sns.histplot(test.Fare, color = 'yellow', alpha = 0.7, ax = axs[0]) \ng1.set_title('Fare distribution training set/test set') \n\n# Standardizing\n# *We will not insert the scaled variables until the preprocessing stage but we need them for the qq-plot\nscaler = StandardScaler()\nz_Fare_train = scaler.fit_transform(train[['Fare']])\nz_Fare_test  = scaler.transform(test[['Fare']])\n\n# Q-Q plot \ng2 = sm.qqplot(z_Fare_train[~np.isnan(z_Fare_train)], line='45', ax = axs[1])\ng2 = sm.qqplot(z_Fare_test[~np.isnan(z_Fare_test)], line='45', markerfacecolor = 'yellow', markeredgecolor='yellow', alpha = 0.5, ax = axs[1])\ng2.suptitle('Q-Q plot Fare training set/test set', x = 0.77, y = 0.93)\n\nlegend = [\n          Line2D([0], [0], marker='s', color='w', label='Test set', markerfacecolor='yellow', markersize=5),\n          Line2D([0], [0], marker='s', color='w', label='Training set', markerfacecolor='royalblue', markersize=5),\n         ]\nfig.legend(handles=legend, fontsize = 'large', bbox_to_anchor=(0.15, 0))\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:13.772837Z","iopub.execute_input":"2022-08-05T09:23:13.773240Z","iopub.status.idle":"2022-08-05T09:23:14.807157Z","shell.execute_reply.started":"2022-08-05T09:23:13.773206Z","shell.execute_reply":"2022-08-05T09:23:14.806051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Fare.median()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:14.808661Z","iopub.execute_input":"2022-08-05T09:23:14.809165Z","iopub.status.idle":"2022-08-05T09:23:14.816996Z","shell.execute_reply.started":"2022-08-05T09:23:14.809132Z","shell.execute_reply":"2022-08-05T09:23:14.815844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Positive Skewed, leptokurtic distribution. \n- Not normal, more like a negative exponential function. (Correspondence with Pclass?) \n- 50% cases below 14 $ \n- Some outliers, no missing values. \n- Similar distribution in the training set and the test set","metadata":{}},{"cell_type":"markdown","source":"<a id=\"pclass\"></a>\n### PClass","metadata":{"tags":[]}},{"cell_type":"markdown","source":"#### Preparing data","metadata":{"tags":[]}},{"cell_type":"code","source":"# Data Conversion \ncat_type = pd.CategoricalDtype(categories=[3, 2, 1], ordered=True)\ntrain['Pclass'] = train.Pclass.astype(cat_type).cat.rename_categories(['Third', 'Second', 'First'])\ntest['Pclass'] = test.Pclass.astype(cat_type).cat.rename_categories(['Third', 'Second', 'First'])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:14.818690Z","iopub.execute_input":"2022-08-05T09:23:14.819172Z","iopub.status.idle":"2022-08-05T09:23:14.833912Z","shell.execute_reply.started":"2022-08-05T09:23:14.819133Z","shell.execute_reply":"2022-08-05T09:23:14.832715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.loc[0:3, 'Pclass'])\nprint(test.loc[0:3, 'Pclass'])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:14.835206Z","iopub.execute_input":"2022-08-05T09:23:14.836334Z","iopub.status.idle":"2022-08-05T09:23:14.859724Z","shell.execute_reply.started":"2022-08-05T09:23:14.836295Z","shell.execute_reply":"2022-08-05T09:23:14.858219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Exploration","metadata":{"tags":[]}},{"cell_type":"code","source":"print('Percentages')\nprint(f'Training set: \\n{train.Pclass.value_counts(normalize = True).mul(100)}', end = '\\n\\n')\nprint(f'Test set: \\n{test.Pclass.value_counts(normalize = True).mul(100)}')\n\nfig, axs = plt.subplots(1,2, figsize = (8, 3), sharey = True)\ng1 = sns.countplot(x = train.Pclass, palette = 'Set2', ax = axs[0])\ng1.set_title('Training set', fontsize = 'medium')\ng2 =  sns.countplot(x = test.Pclass, palette = 'Set1', alpha = 0.5, ax = axs[1])\ng2.set_title('Test set', fontsize = 'medium')\ng2.set(ylim=(0, 500))\nfig.suptitle('Passengers per class', fontsize = 'medium', x = 0.52, y = 1.05)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:14.861655Z","iopub.execute_input":"2022-08-05T09:23:14.862806Z","iopub.status.idle":"2022-08-05T09:23:15.176133Z","shell.execute_reply.started":"2022-08-05T09:23:14.862763Z","shell.execute_reply":"2022-08-05T09:23:15.174497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"sex\"></a>\n### Sex ","metadata":{"tags":[]}},{"cell_type":"markdown","source":"#### Preparing data","metadata":{"tags":[]}},{"cell_type":"code","source":"# Data conversion\ntrain['Sex'] = train.Sex.astype(\"category\")\ntest['Sex'] = test.Sex.astype(\"category\")","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.178400Z","iopub.execute_input":"2022-08-05T09:23:15.178961Z","iopub.status.idle":"2022-08-05T09:23:15.192691Z","shell.execute_reply.started":"2022-08-05T09:23:15.178901Z","shell.execute_reply":"2022-08-05T09:23:15.190510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Exploration","metadata":{"tags":[]}},{"cell_type":"code","source":"print('Percentages')\nprint(f'Training set: \\n{train.Sex.value_counts(normalize = True).mul(100)}', end = '\\n\\n')\nprint(f'Test set: \\n{test.Sex.value_counts(normalize = True).mul(100)}')\n\nfig, axs = plt.subplots(1,2, figsize = (8, 3))\ng1 = sns.countplot(x = train.Sex, palette = 'Set2', ax = axs[0])\ng1.set_title('Training set', fontsize = 'medium')\ng2 =  sns.countplot(x = test.Sex, palette = 'Set1', alpha = 0.5, ax = axs[1])\ng2.set_title('Test set', fontsize = 'medium')\ng2.set(ylim=(0, 600))\nfig.suptitle('Passengers per sex', fontsize = 'medium', x = 0.52, y = 1.05)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.194612Z","iopub.execute_input":"2022-08-05T09:23:15.196622Z","iopub.status.idle":"2022-08-05T09:23:15.511936Z","shell.execute_reply.started":"2022-08-05T09:23:15.196563Z","shell.execute_reply":"2022-08-05T09:23:15.510582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"embarked\"></a>\n### Embarked","metadata":{"tags":[]}},{"cell_type":"markdown","source":"Let's encode Embarked categories with the original name instead of the initial for friendliness. ","metadata":{}},{"cell_type":"markdown","source":"#### Preparing data","metadata":{"tags":[]}},{"cell_type":"code","source":"train['Embarked'] = train.Embarked.astype(\"category\").cat.rename_categories(['Cherbourg', 'Queenstown', 'Southampton'])\ntest['Embarked'] = test.Embarked.astype(\"category\").cat.rename_categories(['Cherbourg', 'Queenstown', 'Southampton'])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.514192Z","iopub.execute_input":"2022-08-05T09:23:15.514587Z","iopub.status.idle":"2022-08-05T09:23:15.526277Z","shell.execute_reply.started":"2022-08-05T09:23:15.514551Z","shell.execute_reply":"2022-08-05T09:23:15.524997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Exploration","metadata":{"tags":[]}},{"cell_type":"code","source":"print('Percentages')\nprint(f'Training set: \\n{train.Embarked.value_counts(normalize = True).mul(100)}', end = '\\n\\n')\nprint(f'Test set: \\n{test.Embarked.value_counts(normalize = True).mul(100)}')\n\nfig, axs = plt.subplots(1,2, figsize = (8, 3))\ng1 = sns.countplot(x = train.Embarked, palette = 'Set2', ax = axs[0])\ng1.set_title('Training set', fontsize = 'medium')\ng2 =  sns.countplot(x = test.Embarked, palette = 'Set1', alpha = 0.5, ax = axs[1])\ng2.set_title('Test set', fontsize = 'medium')\ng2.set(ylim=(0, 680))\nfig.suptitle('Passengers per Port', x = 0.52, y = 1.05, fontsize = 'medium') ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.528235Z","iopub.execute_input":"2022-08-05T09:23:15.528616Z","iopub.status.idle":"2022-08-05T09:23:15.841386Z","shell.execute_reply.started":"2022-08-05T09:23:15.528584Z","shell.execute_reply":"2022-08-05T09:23:15.840311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"parchsib\"></a>\n### Parch and SibSp","metadata":{"tags":[]}},{"cell_type":"markdown","source":"#### Exploration","metadata":{"tags":[]}},{"cell_type":"code","source":"print(train.Parch.unique())\nprint(train.SibSp.unique())\n\nprint(train.Parch.value_counts())\nprint(train.SibSp.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.843424Z","iopub.execute_input":"2022-08-05T09:23:15.844160Z","iopub.status.idle":"2022-08-05T09:23:15.853451Z","shell.execute_reply.started":"2022-08-05T09:23:15.844121Z","shell.execute_reply":"2022-08-05T09:23:15.852186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test.Parch.unique())\nprint(test.SibSp.unique())\n\nprint(test.Parch.value_counts())\nprint(test.SibSp.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.854795Z","iopub.execute_input":"2022-08-05T09:23:15.855692Z","iopub.status.idle":"2022-08-05T09:23:15.867925Z","shell.execute_reply.started":"2022-08-05T09:23:15.855653Z","shell.execute_reply":"2022-08-05T09:23:15.866070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Creating New features","metadata":{"tags":[]}},{"cell_type":"code","source":"# Number of relatives on board\ntrain['nRelatives'] = train['SibSp'] + train['Parch'] \ntest['nRelatives'] = test['SibSp'] + test['Parch'] \n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.869389Z","iopub.execute_input":"2022-08-05T09:23:15.870430Z","iopub.status.idle":"2022-08-05T09:23:15.883508Z","shell.execute_reply.started":"2022-08-05T09:23:15.870381Z","shell.execute_reply":"2022-08-05T09:23:15.882516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Accompanied: 0 = No, 1 = Yes \n\n# Training \narr = np.zeros(len(train['nRelatives']))\nidx = [i for i in range(len(train['nRelatives'])) if train.loc[train.index[i],'nRelatives'] != 0] \narr[idx] = 1\n\ntrain['Accompanied'] = pd.Series(arr, dtype = 'int64', index = train.index)\n\n# Test\narr2 = np.zeros(len(test['nRelatives']))\nidx = [i for i in range(len(test['nRelatives'])) if test.loc[test.index[i],'nRelatives'] != 0] \narr2[idx] = 1\ntest['Accompanied'] = pd.Series(arr2, dtype = 'int64', index = test.index)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.885013Z","iopub.execute_input":"2022-08-05T09:23:15.886212Z","iopub.status.idle":"2022-08-05T09:23:15.921055Z","shell.execute_reply.started":"2022-08-05T09:23:15.886163Z","shell.execute_reply":"2022-08-05T09:23:15.919429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training set\\n',pd.DataFrame({'Count': train.Accompanied.value_counts(), 'Percentage': train.Accompanied.value_counts(normalize = True).mul(100)}))\nprint('Test set\\n', pd.DataFrame({'Count': test.Accompanied.value_counts(), 'Percentage': test.Accompanied.value_counts(normalize = True).mul(100)}))\n\n\nprint('Training set\\n',pd.DataFrame({'Count': train.nRelatives.value_counts(), 'Percentage': train.nRelatives.value_counts(normalize = True).mul(100)}))\nprint('Test set\\n', pd.DataFrame({'Count': test.nRelatives.value_counts(), 'Percentage': test.nRelatives.value_counts(normalize = True).mul(100)}))\n\n\ng= sns.countplot(x = train.nRelatives, color = 'royalblue')\ng = sns.countplot(x = test.nRelatives, color = 'yellow', alpha =0.9)\ng.set_title('Number of Relatives on board (overlayed sets)')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:15.925971Z","iopub.execute_input":"2022-08-05T09:23:15.926735Z","iopub.status.idle":"2022-08-05T09:23:16.221874Z","shell.execute_reply.started":"2022-08-05T09:23:15.926689Z","shell.execute_reply":"2022-08-05T09:23:16.220522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The majority of passengers (around 60%) don't have a relative on board","metadata":{}},{"cell_type":"markdown","source":"<a id=\"tickcab\"></a>\n### Ticket and Cabin","metadata":{"tags":[]}},{"cell_type":"markdown","source":"#### Exploration","metadata":{}},{"cell_type":"code","source":"print(f'Number of unique Cabins: {train.Cabin.dropna().nunique()}')\nprint(f'Number of duplicated Cabins: {len(train.Cabin.dropna()) - train.Cabin.dropna().nunique()}')\n\ntrain[train.Cabin.duplicated() & train.Cabin.notnull()].Accompanied.value_counts() ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:16.223142Z","iopub.execute_input":"2022-08-05T09:23:16.223507Z","iopub.status.idle":"2022-08-05T09:23:16.240304Z","shell.execute_reply.started":"2022-08-05T09:23:16.223476Z","shell.execute_reply":"2022-08-05T09:23:16.239001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of unique tickets: {train.Ticket.nunique()}')\nprint(f'Number of duplicated tickets: {len(train.Ticket) - train.Ticket.nunique()}')\n\ntrain.loc[train.Ticket.duplicated()].Accompanied.value_counts() # \n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:16.242608Z","iopub.execute_input":"2022-08-05T09:23:16.243529Z","iopub.status.idle":"2022-08-05T09:23:16.258566Z","shell.execute_reply.started":"2022-08-05T09:23:16.243482Z","shell.execute_reply":"2022-08-05T09:23:16.256851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = set(train[train.Cabin.duplicated() & train.Cabin.notnull()].index) # Index of Duplicated Cabins\nb = set(train.loc[train.Ticket.duplicated()].index) # Index of duplicated tickets\n\nprint('Number of duplicated Cabins:', len(a),'\\nNumber of duplicated Tickets:', len(b), '\\nNumber of common cases:', len(a.intersection(b)))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:16.259759Z","iopub.execute_input":"2022-08-05T09:23:16.260615Z","iopub.status.idle":"2022-08-05T09:23:16.272340Z","shell.execute_reply.started":"2022-08-05T09:23:16.260579Z","shell.execute_reply":"2022-08-05T09:23:16.270906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The passengers that share cabin usually have relatives traveling on board. \n- The passengers having the same ticket number usually have relatives on board. \n- The majority of passengers that share a cabin also have the same ticket number as someone. \n- This could mean that these people traveled together in the same cabin. ","metadata":{}},{"cell_type":"markdown","source":"#### Creating new features","metadata":{"tags":[]}},{"cell_type":"code","source":"# Missing indicator for Cabin\ntrain['Miss_Cabin'] = train.Cabin.map(pd.isnull).astype('int64')\ntest['Miss_Cabin'] = test.Cabin.map(pd.isnull).astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:16.274482Z","iopub.execute_input":"2022-08-05T09:23:16.274993Z","iopub.status.idle":"2022-08-05T09:23:16.290715Z","shell.execute_reply.started":"2022-08-05T09:23:16.274947Z","shell.execute_reply":"2022-08-05T09:23:16.289486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"export_sec\"></a>\n## Exporting","metadata":{"tags":[]}},{"cell_type":"markdown","source":"### Preparing data for exportation ","metadata":{}},{"cell_type":"code","source":"train = train.drop(columns = ['Name', 'Cabin', 'Ticket']) # Dropping columns that we won't be using\ntrain = train.reindex(columns = ['Survived', 'Sex', 'Pclass', 'Embarked', 'Age', 'Fare', 'Miss_Cabin', \n                                 'Accompanied', 'nRelatives', 'SibSp', 'Parch']) # Reordering Columns \ndisplay(train.head())\n\n\ntest = test.drop(columns = ['Name', 'Cabin', 'Ticket']) # Dropping columns that we won't be using \ntest = test.reindex(columns = ['Sex', 'Pclass', 'Embarked', 'Age', 'Fare', 'Miss_Cabin', \n                                 'Accompanied', 'nRelatives', 'SibSp', 'Parch']) # Reordering Columns \ndisplay(test.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:16.292469Z","iopub.execute_input":"2022-08-05T09:23:16.293115Z","iopub.status.idle":"2022-08-05T09:23:16.335810Z","shell.execute_reply.started":"2022-08-05T09:23:16.293037Z","shell.execute_reply":"2022-08-05T09:23:16.334521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exportation\n#train.to_csv('EDA_Univar/train_univar.csv')\n#test.to_csv('EDA_Univar/train_univar.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T09:23:16.337330Z","iopub.execute_input":"2022-08-05T09:23:16.337799Z","iopub.status.idle":"2022-08-05T09:23:16.373861Z","shell.execute_reply.started":"2022-08-05T09:23:16.337749Z","shell.execute_reply":"2022-08-05T09:23:16.372559Z"},"trusted":true},"execution_count":null,"outputs":[]}]}