{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns # data visualization\nimport matplotlib.pyplot as plt # data visualization\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T17:10:27.008561Z","iopub.execute_input":"2022-07-11T17:10:27.009305Z","iopub.status.idle":"2022-07-11T17:10:28.341494Z","shell.execute_reply.started":"2022-07-11T17:10:27.009205Z","shell.execute_reply":"2022-07-11T17:10:28.339958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA along with feature engineering","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv') # reading the data\ntrain.head() # looking at the first 5 rows","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:10:28.344126Z","iopub.execute_input":"2022-07-11T17:10:28.344973Z","iopub.status.idle":"2022-07-11T17:10:28.453031Z","shell.execute_reply.started":"2022-07-11T17:10:28.344922Z","shell.execute_reply":"2022-07-11T17:10:28.452082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info() # using the info method to look at the number of values and data type for each column","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:16.755951Z","iopub.execute_input":"2022-07-11T17:17:16.756438Z","iopub.status.idle":"2022-07-11T17:17:16.796399Z","shell.execute_reply.started":"2022-07-11T17:17:16.756404Z","shell.execute_reply":"2022-07-11T17:17:16.795527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Cabin has a lot of unique values. Using it as a categorical variable won't be wise. Instead, lets do feature engineering and create some features out of it.","metadata":{}},{"cell_type":"code","source":"train['Floor'] = train['Cabin'].str[:1] # Feature engineering a floor feature from the Cabin feature (taking the first value of the Cabin string)\ntrain['CabinPort'] = train['Cabin'].str[-1:] # Feature engineering a floor feature from the Cabin feature (taking the last value of the Cabin string)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:17.173401Z","iopub.execute_input":"2022-07-11T17:17:17.174181Z","iopub.status.idle":"2022-07-11T17:17:17.189746Z","shell.execute_reply.started":"2022-07-11T17:17:17.174140Z","shell.execute_reply":"2022-07-11T17:17:17.188460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# printing the number of null and unique values in every feature\nall_cols = train.columns \nfor col in all_cols:\n    print(f'{col} has {train[col].isna().sum()} null values and {train[col].nunique()} unique values')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:17.363539Z","iopub.execute_input":"2022-07-11T17:17:17.364241Z","iopub.status.idle":"2022-07-11T17:17:17.404046Z","shell.execute_reply.started":"2022-07-11T17:17:17.364205Z","shell.execute_reply":"2022-07-11T17:17:17.402789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(['Name', 'Cabin'], axis=1) # Dropping Name and Cabin columns as they aren't useful","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:17.560349Z","iopub.execute_input":"2022-07-11T17:17:17.560774Z","iopub.status.idle":"2022-07-11T17:17:17.571915Z","shell.execute_reply.started":"2022-07-11T17:17:17.560741Z","shell.execute_reply":"2022-07-11T17:17:17.570816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making two lists of Categorical and Numerical Columns\ncat_cols = ['HomePlanet', 'CryoSleep', 'Destination', 'VIP', 'Floor', 'CabinPort']\nnum_cols = ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck', 'Age']","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:17.738923Z","iopub.execute_input":"2022-07-11T17:17:17.739587Z","iopub.status.idle":"2022-07-11T17:17:17.744979Z","shell.execute_reply.started":"2022-07-11T17:17:17.739542Z","shell.execute_reply":"2022-07-11T17:17:17.744063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Handling Missing Values using EDA\n##### Test data is 1/3rd the size of total data, train data is 2/3rd the size of total data. Such a split means that we somehow have to retain as much data as we can. Sadly, I have to remove null values from categorical columns as I want to use frequency encoding for that which doesn't take nan values. I can replace nan values with some string and then use it in frequency encoding but then the meaning of the data would change so lets not do that.","metadata":{}},{"cell_type":"code","source":"# Dropping null values in categorical columns\ntrain.dropna(subset=cat_cols, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:18.084021Z","iopub.execute_input":"2022-07-11T17:17:18.084655Z","iopub.status.idle":"2022-07-11T17:17:18.101576Z","shell.execute_reply.started":"2022-07-11T17:17:18.084618Z","shell.execute_reply":"2022-07-11T17:17:18.100114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting Stripplots for numerical columns with the target column as hue and CabinPort on the x-axis.\n# Useful for finding the relation between CabinPort and numerical columns\nfig, axes = plt.subplots(2, 3, figsize=(25,10))\nfor i, col in zip(range(6), num_cols):\n    sns.stripplot(ax=axes[i//3][i%3], x='CabinPort', y=col, data=train, palette='GnBu', hue='Transported')\n    axes[i//3][i%3].set_title(f'{col} Stripplot')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:18.241621Z","iopub.execute_input":"2022-07-11T17:17:18.242806Z","iopub.status.idle":"2022-07-11T17:17:20.205863Z","shell.execute_reply.started":"2022-07-11T17:17:18.242738Z","shell.execute_reply":"2022-07-11T17:17:20.204356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Stripplots based on CabinPort provide valuable insights. Each numerical column has some relation to it and so it is better to impute the null values in it using the mean for specific CabinPort.","metadata":{}},{"cell_type":"code","source":"# So here we are replacing the missing values in the numerical columns. We are calculating the mean of each numerical column\n# separately for the rows that have CabinPort as 'P' and the rows that have CabinPort as 'S'\nfor col in num_cols:\n    p_mean = train[col].loc[train['CabinPort'] == 'P'].mean()\n    s_mean = train[col].loc[train['CabinPort'] == 'S'].mean()\n    train.loc[(train['CabinPort'] == 'P') & (train[col].isna()), col] = p_mean\n    train.loc[(train['CabinPort'] == 'S') & (train[col].isna()), col] = s_mean\n    \n# Printing the null values of numerical columns\nfor col in num_cols:\n    print(f'{col} has {train[col].isna().sum()} null values')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:20.208044Z","iopub.execute_input":"2022-07-11T17:17:20.208412Z","iopub.status.idle":"2022-07-11T17:17:20.273050Z","shell.execute_reply.started":"2022-07-11T17:17:20.208381Z","shell.execute_reply":"2022-07-11T17:17:20.271741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Outliers","metadata":{}},{"cell_type":"code","source":"# Now we will be using Box and Whiskers Plot to check for outliers\nfig, axes = plt.subplots(2, 3, figsize=(25,10))\nfor i, col in zip(range(6), num_cols):\n    sns.boxplot(ax=axes[i//3][i%3], x='Transported', y=col, data=train, palette='GnBu', hue='Transported')\n    axes[i//3][i%3].set_title(f'{col} Boxplot')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:20.274598Z","iopub.execute_input":"2022-07-11T17:17:20.276315Z","iopub.status.idle":"2022-07-11T17:17:21.323757Z","shell.execute_reply.started":"2022-07-11T17:17:20.276267Z","shell.execute_reply":"2022-07-11T17:17:21.322498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Hmmm...the plots aren't clear. The reason: Outliers! Let's remove them, I don't like outliers. Box and Whisker plots are a famous method for seeing outliers.","metadata":{}},{"cell_type":"markdown","source":"![](https://www150.statcan.gc.ca/edu/power-pouvoir/fig/fig04-5-2-1-eng.png)","metadata":{}},{"cell_type":"code","source":"# This function will return the upper and lower limits of the specified numerical column\ndef outlier_limits(df, col_name, q1 = 0.25, q3 = 0.75):\n    quartile1 = df[col_name].quantile(q1)\n    quartile3 = df[col_name].quantile(q3)\n    interquantile_range = quartile3 - quartile1\n    up_limit = quartile3 + 1.5 * interquantile_range\n    low_limit = quartile1 - 1.5 * interquantile_range\n    return low_limit, up_limit\n\n# This function returns the number of outliers in the specified numerical column\ndef no_of_outliers(df, variable, q1 = 0.25, q3 = 0.75):\n    low_limit, up_limit = outlier_limits(df, variable, q1 = q1, q3 = q3)\n    return df.loc[(df[variable] < low_limit) | (df[variable] > up_limit), variable].count()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:21.326569Z","iopub.execute_input":"2022-07-11T17:17:21.326920Z","iopub.status.idle":"2022-07-11T17:17:21.335562Z","shell.execute_reply.started":"2022-07-11T17:17:21.326890Z","shell.execute_reply":"2022-07-11T17:17:21.334544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Here we are using our defined function in the above cell to see how many outliers are there in the numerical columns\nfor col in num_cols:\n    count = no_of_outliers(train, col)\n    print(f'{col} has {count} outliers')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:21.336868Z","iopub.execute_input":"2022-07-11T17:17:21.337555Z","iopub.status.idle":"2022-07-11T17:17:21.370613Z","shell.execute_reply.started":"2022-07-11T17:17:21.337521Z","shell.execute_reply":"2022-07-11T17:17:21.369360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### The data is filled with outliers! Since the number of outliers is a lot, we will not remove them, but replace them with the upper and lower limits.","metadata":{}},{"cell_type":"code","source":"# This function uses the upper and lower limits generated to replace the outliers with lower limit or upper limit, whichever applies\ndef replace_with_limits(df, variable, q1 = 0.25, q3 = 0.75):\n    low_limit, up_limit = outlier_limits(df, variable, q1 = q1, q3 = q3)\n    df.loc[(df[variable] < low_limit), variable] = low_limit\n    df.loc[(df[variable] > up_limit), variable] = up_limit\n\n# Here we are using our defined function to replace the outliers with the upper and lower limits\nfor col in num_cols:\n    replace_with_limits(train, col)\n    count = no_of_outliers(train, col)\n    print(f'{col} has {count} outliers')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:21.371817Z","iopub.execute_input":"2022-07-11T17:17:21.372979Z","iopub.status.idle":"2022-07-11T17:17:21.414113Z","shell.execute_reply.started":"2022-07-11T17:17:21.372944Z","shell.execute_reply":"2022-07-11T17:17:21.412900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### We have successfully replaced all the outliers! Removing so many outliers would have reduced the size of data by more than 20% so it was better to replace it. ","metadata":{}},{"cell_type":"code","source":"# Now we are making some Countplots with hue as our target column to find some valuable insights\nfig, axes = plt.subplots(2, 3, figsize=(25,10))\nfor i, col in zip(range(6), num_cols):\n    sns.histplot(ax=axes[i//3][i%3], x=col, data=train, palette='GnBu', hue='Transported', bins=5, multiple='dodge')\n    axes[i//3][i%3].set_title(f'{col} Countplot')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:21.415876Z","iopub.execute_input":"2022-07-11T17:17:21.416666Z","iopub.status.idle":"2022-07-11T17:17:22.822546Z","shell.execute_reply.started":"2022-07-11T17:17:21.416621Z","shell.execute_reply":"2022-07-11T17:17:22.821097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Well, either people spent a lot on the spaceship, or very little, there is no in-between. But don't forget, there still is the possibility that maybe some spent more on shopping and little on spa. One insight that we gain from here is surprisingly, a lot of people who spent less got transported.","metadata":{}},{"cell_type":"markdown","source":"### Data Preprocessing","metadata":{}},{"cell_type":"code","source":"encoded_train = train.copy() # copying the data to a new variable","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:22.823907Z","iopub.execute_input":"2022-07-11T17:17:22.824275Z","iopub.status.idle":"2022-07-11T17:17:22.831763Z","shell.execute_reply.started":"2022-07-11T17:17:22.824245Z","shell.execute_reply":"2022-07-11T17:17:22.830269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Frequency Encoding: More the frequency of the label, higher the value of encoded label.","metadata":{}},{"cell_type":"code","source":"#frequencyencoding\nfor col in ['HomePlanet', 'CryoSleep', 'Destination', 'Floor']:\n    encoded_col = (encoded_train.groupby(col).size()) / len(encoded_train)\n    encoded_train[col] = encoded_train[col].apply(lambda x : encoded_col[x])\n    \nencoded_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:22.834985Z","iopub.execute_input":"2022-07-11T17:17:22.835404Z","iopub.status.idle":"2022-07-11T17:17:23.050753Z","shell.execute_reply.started":"2022-07-11T17:17:22.835370Z","shell.execute_reply":"2022-07-11T17:17:23.049731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Label Encoder: Converts string into numbered data (int).","metadata":{}},{"cell_type":"code","source":"# Here we will be using sklearn's LabelEncoder to label encode the VIP and Transported columns\nfrom sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\nencoded_vip = le.fit_transform(encoded_train['VIP'])\nencoded_transported = le.fit_transform(encoded_train['Transported'])\nencoded_train['VIP'] = encoded_vip\nencoded_train['Transported'] = encoded_transported\nencoded_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:23.052094Z","iopub.execute_input":"2022-07-11T17:17:23.052427Z","iopub.status.idle":"2022-07-11T17:17:23.154062Z","shell.execute_reply.started":"2022-07-11T17:17:23.052399Z","shell.execute_reply":"2022-07-11T17:17:23.152714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Correlation Heatmap \n##### A correlation heatmap tells us how correlated two variables are.\n##### This is why we encoded our data. Without encoded data, we wouldn't have been able to make a pairplot.","metadata":{}},{"cell_type":"code","source":"# This is why we encoded the categorical columns. Spearman correlation does consider categorical variables\n# to find the correlation between two variables unlike pearson correlation\ncorr = encoded_train[1:].corr(method = 'spearman')\nplt.figure(figsize=(20,6))\nsns.heatmap(corr, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Spearman Correlation Heatmap')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:23.155766Z","iopub.execute_input":"2022-07-11T17:17:23.156151Z","iopub.status.idle":"2022-07-11T17:17:24.053883Z","shell.execute_reply.started":"2022-07-11T17:17:23.156117Z","shell.execute_reply":"2022-07-11T17:17:24.052683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pairplot","metadata":{}},{"cell_type":"code","source":"# We are using seaborn's pairplot to make many different plots to find some insights\nfigure = plt.figure(figsize=(20,30))\nsns.pairplot(data=train, vars=num_cols, hue='Transported', palette='GnBu')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:24.056779Z","iopub.execute_input":"2022-07-11T17:17:24.057951Z","iopub.status.idle":"2022-07-11T17:17:45.685250Z","shell.execute_reply.started":"2022-07-11T17:17:24.057898Z","shell.execute_reply":"2022-07-11T17:17:45.683745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### The reason for such plots is that we had replaced outliers with upper and lower limits.","metadata":{}},{"cell_type":"markdown","source":"### More EDA","metadata":{}},{"cell_type":"code","source":"# Some scatterplots with Age as on the x axis to find some insights\nfig, axes = plt.subplots(1, 3, figsize=(20,7))\nfor i, col in zip(range(3), num_cols[0:3]):\n    sns.scatterplot(ax=axes[i], x='Age', y=col, hue=\"Transported\", style=\"VIP\", data=encoded_train, palette=\"GnBu\")\n\nfig, axes = plt.subplots(1, 2, figsize=(20,7))\nfor i, col in zip(range(2), num_cols[0:5]):\n    sns.scatterplot(ax=axes[i], x='Age', y=col, hue=\"Transported\", style=\"VIP\", data=encoded_train, palette=\"GnBu\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:45.686892Z","iopub.execute_input":"2022-07-11T17:17:45.687380Z","iopub.status.idle":"2022-07-11T17:17:49.415919Z","shell.execute_reply.started":"2022-07-11T17:17:45.687338Z","shell.execute_reply":"2022-07-11T17:17:49.414764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Even many VIPS weren't transported. One insight that we get from this is a lot of VIPS are in the upper and lower limits of the data, and not in between.","metadata":{}},{"cell_type":"code","source":"# Plotting some Stripplots to find more insights\nfig, axes = plt.subplots(2, 3, figsize=(25,10))\nfor i, col in zip(range(6), num_cols):\n    sns.stripplot(ax=axes[i//3][i%3], x='Transported', y=col, data=train, palette='GnBu', hue='VIP', jitter=True)\n    axes[i//3][i%3].set_title(f'{col} Stripplot')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:49.418353Z","iopub.execute_input":"2022-07-11T17:17:49.419411Z","iopub.status.idle":"2022-07-11T17:17:51.112536Z","shell.execute_reply.started":"2022-07-11T17:17:49.419365Z","shell.execute_reply":"2022-07-11T17:17:51.111037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting a countplot with VIP on the x-axis to verify the finding that a lot of VIPs weren't transported\nsns.histplot(x='Transported', data=train, palette='GnBu', hue='VIP', bins=2, multiple='dodge')\nplt.title(f'VIPs Countplot')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T17:17:51.114407Z","iopub.execute_input":"2022-07-11T17:17:51.115734Z","iopub.status.idle":"2022-07-11T17:17:51.363922Z","shell.execute_reply.started":"2022-07-11T17:17:51.115688Z","shell.execute_reply":"2022-07-11T17:17:51.362983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Yup, a lot of VIPs weren't transported.\n##### As you can see, we have gained a lot of insights from EDA and have also done some feature engineering. That is why these steps are important before Machine Learning. This is where we end our EDA. Please upvote this notebook if you like it. It motivates me to create more and more such notebooks. Cheers🍻","metadata":{}}]}