{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T02:51:36.596510Z","iopub.execute_input":"2022-08-03T02:51:36.596977Z","iopub.status.idle":"2022-08-03T02:51:36.606110Z","shell.execute_reply.started":"2022-08-03T02:51:36.596938Z","shell.execute_reply":"2022-08-03T02:51:36.605067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import warnings filter\nfrom warnings import simplefilter\n# ignore all future warnings\nsimplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:51:43.220646Z","iopub.execute_input":"2022-08-03T02:51:43.221673Z","iopub.status.idle":"2022-08-03T02:51:43.227643Z","shell.execute_reply.started":"2022-08-03T02:51:43.221619Z","shell.execute_reply":"2022-08-03T02:51:43.226376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:51:41.804622Z","iopub.execute_input":"2022-08-03T02:51:41.805601Z","iopub.status.idle":"2022-08-03T02:51:41.810518Z","shell.execute_reply.started":"2022-08-03T02:51:41.805542Z","shell.execute_reply":"2022-08-03T02:51:41.809559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nfrom matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:51:39.864701Z","iopub.execute_input":"2022-08-03T02:51:39.865109Z","iopub.status.idle":"2022-08-03T02:51:40.557429Z","shell.execute_reply.started":"2022-08-03T02:51:39.865074Z","shell.execute_reply":"2022-08-03T02:51:40.556167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:51:44.592459Z","iopub.execute_input":"2022-08-03T02:51:44.593413Z","iopub.status.idle":"2022-08-03T02:51:44.615587Z","shell.execute_reply.started":"2022-08-03T02:51:44.593360Z","shell.execute_reply":"2022-08-03T02:51:44.614669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:51:48.404271Z","iopub.execute_input":"2022-08-03T02:51:48.404746Z","iopub.status.idle":"2022-08-03T02:51:48.415580Z","shell.execute_reply.started":"2022-08-03T02:51:48.404705Z","shell.execute_reply":"2022-08-03T02:51:48.414178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df.isnull(), cbar = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:51:50.420180Z","iopub.execute_input":"2022-08-03T02:51:50.420965Z","iopub.status.idle":"2022-08-03T02:51:50.754519Z","shell.execute_reply.started":"2022-08-03T02:51:50.420888Z","shell.execute_reply":"2022-08-03T02:51:50.753221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:51:52.766388Z","iopub.execute_input":"2022-08-03T02:51:52.766804Z","iopub.status.idle":"2022-08-03T02:51:52.776576Z","shell.execute_reply.started":"2022-08-03T02:51:52.766772Z","shell.execute_reply":"2022-08-03T02:51:52.775322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x = df.columns, y = df.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:51:54.722921Z","iopub.execute_input":"2022-08-03T02:51:54.723345Z","iopub.status.idle":"2022-08-03T02:51:54.984283Z","shell.execute_reply.started":"2022-08-03T02:51:54.723313Z","shell.execute_reply":"2022-08-03T02:51:54.982978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PassangerID, Name, Ticket, Fare & Cabin can be dropped as :\n  \n- PassangerID and Name will not have any impact on survival \n\n- PClass already defines the class in which passanger is traveling. So, no need to consider ticket.\n\n- Cabin has around 80% null values\n    ","metadata":{}},{"cell_type":"code","source":"drop_cols = ['PassengerId', 'Name', 'Ticket', 'Cabin']","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:52:10.036010Z","iopub.execute_input":"2022-08-03T02:52:10.036482Z","iopub.status.idle":"2022-08-03T02:52:10.042168Z","shell.execute_reply.started":"2022-08-03T02:52:10.036445Z","shell.execute_reply":"2022-08-03T02:52:10.040927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.loc[:, ~df.columns.isin(drop_cols)]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:52:11.443173Z","iopub.execute_input":"2022-08-03T02:52:11.443945Z","iopub.status.idle":"2022-08-03T02:52:11.454577Z","shell.execute_reply.started":"2022-08-03T02:52:11.443878Z","shell.execute_reply":"2022-08-03T02:52:11.453404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:52:12.630779Z","iopub.execute_input":"2022-08-03T02:52:12.631971Z","iopub.status.idle":"2022-08-03T02:52:12.654013Z","shell.execute_reply.started":"2022-08-03T02:52:12.631891Z","shell.execute_reply":"2022-08-03T02:52:12.652346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(df, hue = 'Survived')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:52:15.187131Z","iopub.execute_input":"2022-08-03T02:52:15.187666Z","iopub.status.idle":"2022-08-03T02:52:22.568697Z","shell.execute_reply.started":"2022-08-03T02:52:15.187626Z","shell.execute_reply":"2022-08-03T02:52:22.567675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical = ['Survived', 'Pclass', 'Age', 'SibSp', 'Parch', 'Fare']\n\ncategories = ['Sex', 'Embarked']","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:52:50.763274Z","iopub.execute_input":"2022-08-03T02:52:50.763767Z","iopub.status.idle":"2022-08-03T02:52:50.770581Z","shell.execute_reply.started":"2022-08-03T02:52:50.763727Z","shell.execute_reply":"2022-08-03T02:52:50.769122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,3, figsize = (20,10))\n\nfor variable, subplot in zip(numerical, ax.flatten()):\n    \n    sns.histplot(df[variable],kde = True, ax = subplot)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:52:54.019027Z","iopub.execute_input":"2022-08-03T02:52:54.019442Z","iopub.status.idle":"2022-08-03T02:52:55.410605Z","shell.execute_reply.started":"2022-08-03T02:52:54.019408Z","shell.execute_reply":"2022-08-03T02:52:55.409631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize = (20, 10))\n\nfor variable, subplot in zip(categories, ax.flatten()):\n    \n    sns.countplot(x = variable, ax = subplot, hue = 'Survived', data = df)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:53:01.584848Z","iopub.execute_input":"2022-08-03T02:53:01.585308Z","iopub.status.idle":"2022-08-03T02:53:01.974770Z","shell.execute_reply.started":"2022-08-03T02:53:01.585272Z","shell.execute_reply":"2022-08-03T02:53:01.973572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,3, figsize = (30, 20))\n\nfor variable, subplots in zip(numerical, ax.flatten()):\n    \n    sns.histplot(x = variable, data = df, hue = 'Survived',kde = True, ax = subplots  )","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:53:04.195681Z","iopub.execute_input":"2022-08-03T02:53:04.196628Z","iopub.status.idle":"2022-08-03T02:53:06.772104Z","shell.execute_reply.started":"2022-08-03T02:53:04.196582Z","shell.execute_reply":"2022-08-03T02:53:06.770692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the plots above,\n\n- Pclass 3 has low rate of survival\n\n- age between 20 and 50 have more values and lesser survival rate than the other age groups\n\n- SibSp (Sibling and spouses) 0 and Parch (Parents and child) 0 have greater count but survival is low\n\n- SibSp, Parch, Fare and Age have some outliers","metadata":{}},{"cell_type":"markdown","source":"Outliers","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1,2, figsize = (20,10))\n\nplt.suptitle('SibSp Outlier Study')\n\nsns.countplot(df.SibSp, ax = axes[0])\nsns.boxplot(df.SibSp, ax = axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:53:37.346200Z","iopub.execute_input":"2022-08-03T02:53:37.348790Z","iopub.status.idle":"2022-08-03T02:53:37.702214Z","shell.execute_reply.started":"2022-08-03T02:53:37.348705Z","shell.execute_reply":"2022-08-03T02:53:37.700015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df['SibSp'] >4, :].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:53:40.216737Z","iopub.execute_input":"2022-08-03T02:53:40.217493Z","iopub.status.idle":"2022-08-03T02:53:40.227582Z","shell.execute_reply.started":"2022-08-03T02:53:40.217440Z","shell.execute_reply":"2022-08-03T02:53:40.226134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,2, figsize = (20,10))\n\nplt.suptitle('Parch Outlier Study')\n\nsns.countplot(df.Parch, ax = axes[0])\nsns.boxplot(df.Parch, ax = axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:53:40.613357Z","iopub.execute_input":"2022-08-03T02:53:40.614377Z","iopub.status.idle":"2022-08-03T02:53:40.917233Z","shell.execute_reply.started":"2022-08-03T02:53:40.614257Z","shell.execute_reply":"2022-08-03T02:53:40.915854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df['Parch'] >2, :].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:53:44.465371Z","iopub.execute_input":"2022-08-03T02:53:44.465852Z","iopub.status.idle":"2022-08-03T02:53:44.475889Z","shell.execute_reply.started":"2022-08-03T02:53:44.465815Z","shell.execute_reply":"2022-08-03T02:53:44.474930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,2, figsize = (20,10))\n\nplt.suptitle('Age Outlier Study')\n\nsns.countplot(df.Age, ax = axes[0])\naxes[0].tick_params(rotation = 90, labelsize = 'large')\nsns.boxplot(df.Age, ax = axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:53:45.317197Z","iopub.execute_input":"2022-08-03T02:53:45.318703Z","iopub.status.idle":"2022-08-03T02:53:46.513506Z","shell.execute_reply.started":"2022-08-03T02:53:45.318649Z","shell.execute_reply":"2022-08-03T02:53:46.512014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df['Age'] >65, :].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:11:50.001457Z","iopub.execute_input":"2022-08-02T09:11:50.002414Z","iopub.status.idle":"2022-08-02T09:11:50.011418Z","shell.execute_reply.started":"2022-08-02T09:11:50.002368Z","shell.execute_reply":"2022-08-02T09:11:50.010289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,2, figsize = (20,10))\n\nplt.suptitle('Fare Outlier Study')\n\nsns.countplot(df.Fare, ax = axes[0])\naxes[0].tick_params(rotation = 90, labelsize = 'large')\nsns.boxplot(df.Fare, ax = axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:54:05.583395Z","iopub.execute_input":"2022-08-03T02:54:05.584503Z","iopub.status.idle":"2022-08-03T02:54:08.865086Z","shell.execute_reply.started":"2022-08-03T02:54:05.584459Z","shell.execute_reply":"2022-08-03T02:54:08.863822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df['Fare'] >200, :].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:55:18.044765Z","iopub.execute_input":"2022-08-03T02:55:18.045219Z","iopub.status.idle":"2022-08-03T02:55:18.055550Z","shell.execute_reply.started":"2022-08-03T02:55:18.045184Z","shell.execute_reply":"2022-08-03T02:55:18.054200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the analysis above, considering all the Parch >2, SibSp >3 and age > 65 as outliers.\n\nAs it comprises of around 6% of the total data, will delete those records.","metadata":{}},{"cell_type":"markdown","source":"# Data Cleanup and Preprocessing\n\n- We have two categorical Values (Sex, Embarked), we can use One hot Encoding for those values.\n\n- Need to take care the null values in age column, we will use KNN imputer for that.\n\n- There are Outliers (Age, Parch, SibSp) detected from the analysis.\n\n- Encode Age, Parch, SibSp based on the frequency of that data.\n\n- We are using Std scaler to scale the data.\n\n- Need to define a function to encode and prepare the test data as per the preprocessing.","metadata":{}},{"cell_type":"code","source":"df = pd.concat([df, pd.get_dummies(df['Embarked'])], axis = 1)\n\ndf.drop('Embarked', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:56:25.721979Z","iopub.execute_input":"2022-08-03T02:56:25.722434Z","iopub.status.idle":"2022-08-03T02:56:25.737128Z","shell.execute_reply.started":"2022-08-03T02:56:25.722394Z","shell.execute_reply":"2022-08-03T02:56:25.735850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df, pd.get_dummies(df['Sex'])], axis = 1)\n\ndf.drop('Sex', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:56:26.318069Z","iopub.execute_input":"2022-08-03T02:56:26.318487Z","iopub.status.idle":"2022-08-03T02:56:26.329745Z","shell.execute_reply.started":"2022-08-03T02:56:26.318454Z","shell.execute_reply":"2022-08-03T02:56:26.328397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df, pd.get_dummies(df['Pclass'])], axis = 1)\n\ndf.drop('Pclass', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:56:26.914552Z","iopub.execute_input":"2022-08-03T02:56:26.915251Z","iopub.status.idle":"2022-08-03T02:56:26.925180Z","shell.execute_reply.started":"2022-08-03T02:56:26.915210Z","shell.execute_reply":"2022-08-03T02:56:26.923596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:56:28.686563Z","iopub.execute_input":"2022-08-03T02:56:28.687835Z","iopub.status.idle":"2022-08-03T02:56:28.705034Z","shell.execute_reply.started":"2022-08-03T02:56:28.687790Z","shell.execute_reply":"2022-08-03T02:56:28.703450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(df['Age'], kde = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:56:35.810248Z","iopub.execute_input":"2022-08-03T02:56:35.810953Z","iopub.status.idle":"2022-08-03T02:56:36.075129Z","shell.execute_reply.started":"2022-08-03T02:56:35.810916Z","shell.execute_reply":"2022-08-03T02:56:36.073881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(df.loc[df.Age>65, :].index, inplace = True)\ndf.drop(df.loc[df.Parch>2].index, inplace = True)\ndf.drop(df.loc[df.SibSp>3].index, inplace = True)\ndf.drop(df.loc[df['Fare'] >200].index, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:56:05.500244Z","iopub.execute_input":"2022-08-03T02:56:05.500710Z","iopub.status.idle":"2022-08-03T02:56:05.515359Z","shell.execute_reply.started":"2022-08-03T02:56:05.500672Z","shell.execute_reply":"2022-08-03T02:56:05.513840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df.loc[:, df.columns != 'Survived']\ny = df.loc[:, df.columns == 'Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:04.021540Z","iopub.execute_input":"2022-08-03T02:57:04.022014Z","iopub.status.idle":"2022-08-03T02:57:04.029695Z","shell.execute_reply.started":"2022-08-03T02:57:04.021975Z","shell.execute_reply":"2022-08-03T02:57:04.028540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors = 3)\n\nx_trans = pd.DataFrame(imputer.fit_transform(x), columns = x.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:08.104782Z","iopub.execute_input":"2022-08-03T02:57:08.105240Z","iopub.status.idle":"2022-08-03T02:57:08.137449Z","shell.execute_reply.started":"2022-08-03T02:57:08.105206Z","shell.execute_reply":"2022-08-03T02:57:08.135768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(x_trans['Age'], kde = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:10.245071Z","iopub.execute_input":"2022-08-03T02:57:10.245473Z","iopub.status.idle":"2022-08-03T02:57:10.521662Z","shell.execute_reply.started":"2022-08-03T02:57:10.245443Z","shell.execute_reply":"2022-08-03T02:57:10.520392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_trans.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:14.253229Z","iopub.execute_input":"2022-08-03T02:57:14.253692Z","iopub.status.idle":"2022-08-03T02:57:14.265159Z","shell.execute_reply.started":"2022-08-03T02:57:14.253655Z","shell.execute_reply":"2022-08-03T02:57:14.263633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Null Values are Handled using KNN Imputer without significant change in distribution","metadata":{}},{"cell_type":"markdown","source":"Handling Outliers\n\nParch >2, SibSp >3 and age > 65","metadata":{}},{"cell_type":"markdown","source":"Scaling the Data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:17.833239Z","iopub.execute_input":"2022-08-03T02:57:17.833755Z","iopub.status.idle":"2022-08-03T02:57:17.839628Z","shell.execute_reply.started":"2022-08-03T02:57:17.833714Z","shell.execute_reply":"2022-08-03T02:57:17.838320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_scaled = pd.DataFrame(scaler.fit_transform(x_trans), columns = x_trans.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:18.481175Z","iopub.execute_input":"2022-08-03T02:57:18.481849Z","iopub.status.idle":"2022-08-03T02:57:18.494204Z","shell.execute_reply.started":"2022-08-03T02:57:18.481791Z","shell.execute_reply":"2022-08-03T02:57:18.492000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize = (20, 20))\n\n\nfor variable, subplots in zip(x_scaled.columns, axes.flatten()):\n    \n    sns.histplot(x_scaled[variable], kde = True, ax = subplots)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:20.492804Z","iopub.execute_input":"2022-08-03T02:57:20.493541Z","iopub.status.idle":"2022-08-03T02:57:22.581992Z","shell.execute_reply.started":"2022-08-03T02:57:20.493491Z","shell.execute_reply":"2022-08-03T02:57:22.580831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Defining a transformation Function\n\nOperations :\n\n\n- Drop PassengerID, Name, Ticket, Fare, cabin\n\n- Imputer to handle null values\n\n- One hot encoding for Pclass, Sex, Embarked\n\n- MinMax Scaling for the rest","metadata":{}},{"cell_type":"code","source":"def df_transform(df):\n    \n    # Dropping columns which are not considered\n    \n    drop_cols = ['PassengerId', 'Name', 'Ticket', 'Cabin']    \n    df = df.loc[:, ~df.columns.isin(drop_cols)]\n    \n    # One hot encoding for Embarked\n    \n    df = pd.concat([df, pd.get_dummies(df['Embarked'])], axis = 1)\n    df.drop('Embarked', axis = 1, inplace = True)\n    \n    # One hot encoding for sex\n    \n    df = pd.concat([df, pd.get_dummies(df['Sex'])], axis = 1)\n    df.drop('Sex', axis = 1, inplace = True)\n    \n    # One hot encoding for Pclass\n    \n    df = pd.concat([df, pd.get_dummies(df['Pclass'])], axis = 1)\n    df.drop('Pclass', axis = 1, inplace = True)\n    \n    # Null Handling using inputer\n    \n    df = pd.DataFrame(imputer.transform(df), columns = df.columns)\n    \n    #Scaling the data\n    \n    df = pd.DataFrame(scaler.transform(df), columns = df.columns)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:44.248429Z","iopub.execute_input":"2022-08-03T02:57:44.248963Z","iopub.status.idle":"2022-08-03T02:57:44.259372Z","shell.execute_reply.started":"2022-08-03T02:57:44.248925Z","shell.execute_reply":"2022-08-03T02:57:44.257977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training and Validation","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\nfrom sklearn.metrics import classification_report\n\nfrom sklearn.model_selection import cross_val_score\n\nfrom sklearn.model_selection import GridSearchCV\n\nfrom sklearn.metrics import precision_score\n\nfrom sklearn.metrics import recall_score","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:47.388187Z","iopub.execute_input":"2022-08-03T02:57:47.388646Z","iopub.status.idle":"2022-08-03T02:57:47.394957Z","shell.execute_reply.started":"2022-08-03T02:57:47.388606Z","shell.execute_reply":"2022-08-03T02:57:47.393835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nLRClassifier = LogisticRegression()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:53.148106Z","iopub.execute_input":"2022-08-03T02:57:53.148495Z","iopub.status.idle":"2022-08-03T02:57:53.154233Z","shell.execute_reply.started":"2022-08-03T02:57:53.148464Z","shell.execute_reply":"2022-08-03T02:57:53.152812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cross_val_score(LRClassifier, x_scaled, y, cv = 5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:57:53.792190Z","iopub.execute_input":"2022-08-03T02:57:53.792604Z","iopub.status.idle":"2022-08-03T02:57:53.891268Z","shell.execute_reply.started":"2022-08-03T02:57:53.792573Z","shell.execute_reply":"2022-08-03T02:57:53.889994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters to optimize\n\nparams = {'penalty': ['l1', 'l2', 'elasticnet'], 'C': np.arange(0.1, 100.1, 0.1)}","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:58:04.264398Z","iopub.execute_input":"2022-08-03T02:58:04.264862Z","iopub.status.idle":"2022-08-03T02:58:04.270577Z","shell.execute_reply.started":"2022-08-03T02:58:04.264823Z","shell.execute_reply":"2022-08-03T02:58:04.269364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GSC_LRClassifier = GridSearchCV(estimator = LRClassifier, param_grid = params)\n\nGSC_LRClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T02:58:12.345613Z","iopub.execute_input":"2022-08-03T02:58:12.346259Z","iopub.status.idle":"2022-08-03T03:00:21.210849Z","shell.execute_reply.started":"2022-08-03T02:58:12.346201Z","shell.execute_reply":"2022-08-03T03:00:21.208999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LRClassifier.set_params(**GSC_LRClassifier.best_params_)\n\ncross_val_score(LRClassifier, x_scaled, y, cv = 5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:08:13.562887Z","iopub.execute_input":"2022-08-03T03:08:13.563320Z","iopub.status.idle":"2022-08-03T03:08:13.676749Z","shell.execute_reply.started":"2022-08-03T03:08:13.563286Z","shell.execute_reply":"2022-08-03T03:08:13.675549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LRClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:08:25.390858Z","iopub.execute_input":"2022-08-03T03:08:25.391366Z","iopub.status.idle":"2022-08-03T03:08:25.456110Z","shell.execute_reply.started":"2022-08-03T03:08:25.391325Z","shell.execute_reply":"2022-08-03T03:08:25.454372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### KNN Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nKNNClassifier = KNeighborsClassifier()\nKNNClassifier.get_params()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:08:30.919695Z","iopub.execute_input":"2022-08-03T03:08:30.920540Z","iopub.status.idle":"2022-08-03T03:08:30.929261Z","shell.execute_reply.started":"2022-08-03T03:08:30.920496Z","shell.execute_reply":"2022-08-03T03:08:30.928005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.average(cross_val_score(KNNClassifier, x_scaled, y))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:08:33.012351Z","iopub.execute_input":"2022-08-03T03:08:33.013512Z","iopub.status.idle":"2022-08-03T03:08:33.094249Z","shell.execute_reply.started":"2022-08-03T03:08:33.013465Z","shell.execute_reply":"2022-08-03T03:08:33.093003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'metric': ['minkowski','euclidean', 'manhattan', 'chebyshev'], 'n_neighbors': np.arange(1, 100, 1)}\n\nGSC_KNNClassifier = GridSearchCV(estimator = KNNClassifier, param_grid = params, cv = 5)\n\nGSC_KNNClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:08:37.638804Z","iopub.execute_input":"2022-08-03T03:08:37.639249Z","iopub.status.idle":"2022-08-03T03:09:09.167193Z","shell.execute_reply.started":"2022-08-03T03:08:37.639211Z","shell.execute_reply":"2022-08-03T03:09:09.165673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KNNClassifier.set_params(**GSC_KNNClassifier.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:09:37.467235Z","iopub.execute_input":"2022-08-03T03:09:37.467689Z","iopub.status.idle":"2022-08-03T03:09:37.477060Z","shell.execute_reply.started":"2022-08-03T03:09:37.467652Z","shell.execute_reply":"2022-08-03T03:09:37.475712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.average(cross_val_score(KNNClassifier, x_scaled, y, cv = 5))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:09:40.303223Z","iopub.execute_input":"2022-08-03T03:09:40.304038Z","iopub.status.idle":"2022-08-03T03:09:40.391099Z","shell.execute_reply.started":"2022-08-03T03:09:40.303998Z","shell.execute_reply":"2022-08-03T03:09:40.389959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KNNClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:09:44.186886Z","iopub.execute_input":"2022-08-03T03:09:44.187852Z","iopub.status.idle":"2022-08-03T03:09:44.199860Z","shell.execute_reply.started":"2022-08-03T03:09:44.187806Z","shell.execute_reply":"2022-08-03T03:09:44.198888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Decision Tree Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\n\nDTClassifier = DecisionTreeClassifier()\n\nnp.average(cross_val_score(DTClassifier, x_scaled, y, cv = 5))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:09:47.894941Z","iopub.execute_input":"2022-08-03T03:09:47.896040Z","iopub.status.idle":"2022-08-03T03:09:47.978988Z","shell.execute_reply.started":"2022-08-03T03:09:47.895989Z","shell.execute_reply":"2022-08-03T03:09:47.977711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DTClassifier.get_params()\n\nparams = {'ccp_alpha': np.arange(0.0, 0.5, 0.01),'criterion': ['gini', 'entropy'],'max_depth': np.arange(2,20,1), 'splitter': ['best', 'random']}\n\nGSC_DTClassifier = GridSearchCV(estimator = DTClassifier, param_grid = params, cv =5)\n\nGSC_DTClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:09:50.759207Z","iopub.execute_input":"2022-08-03T03:09:50.759629Z","iopub.status.idle":"2022-08-03T03:11:53.818108Z","shell.execute_reply.started":"2022-08-03T03:09:50.759596Z","shell.execute_reply":"2022-08-03T03:11:53.816767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DTClassifier.set_params(**GSC_DTClassifier.best_params_)\n\nnp.average(cross_val_score(DTClassifier, x_scaled, y, cv = 5))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:12:14.750561Z","iopub.execute_input":"2022-08-03T03:12:14.751045Z","iopub.status.idle":"2022-08-03T03:12:14.796801Z","shell.execute_reply.started":"2022-08-03T03:12:14.751004Z","shell.execute_reply":"2022-08-03T03:12:14.795621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DTClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:12:19.442729Z","iopub.execute_input":"2022-08-03T03:12:19.443187Z","iopub.status.idle":"2022-08-03T03:12:19.459679Z","shell.execute_reply.started":"2022-08-03T03:12:19.443150Z","shell.execute_reply":"2022-08-03T03:12:19.458560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nRFClassifier = RandomForestClassifier()\n\nnp.average(cross_val_score(RFClassifier, x_scaled, y))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:12:20.962772Z","iopub.execute_input":"2022-08-03T03:12:20.963596Z","iopub.status.idle":"2022-08-03T03:12:22.071927Z","shell.execute_reply.started":"2022-08-03T03:12:20.963553Z","shell.execute_reply":"2022-08-03T03:12:22.070635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFClassifier.get_params()\n\nparams = {'criterion': ['gini', 'entropy'],'max_depth': np.arange(2,12,1), 'n_estimators' : np.arange(2, 101, 1)}\n\nGSC_RFClassifier = GridSearchCV(estimator = RFClassifier, param_grid = params, cv =5)\n\nGSC_RFClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:12:24.238692Z","iopub.execute_input":"2022-08-03T03:12:24.239183Z","iopub.status.idle":"2022-08-03T03:29:04.418353Z","shell.execute_reply.started":"2022-08-03T03:12:24.239145Z","shell.execute_reply":"2022-08-03T03:29:04.417188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GSC_RFClassifier.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:29:30.371128Z","iopub.execute_input":"2022-08-03T03:29:30.371563Z","iopub.status.idle":"2022-08-03T03:29:30.379709Z","shell.execute_reply.started":"2022-08-03T03:29:30.371529Z","shell.execute_reply":"2022-08-03T03:29:30.378696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFClassifier.set_params(**GSC_RFClassifier.best_params_)\n\nnp.average(cross_val_score(RFClassifier, x_scaled, y, cv = 5))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:29:31.444187Z","iopub.execute_input":"2022-08-03T03:29:31.444885Z","iopub.status.idle":"2022-08-03T03:29:31.711696Z","shell.execute_reply.started":"2022-08-03T03:29:31.444849Z","shell.execute_reply":"2022-08-03T03:29:31.710428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RFClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:29:33.446265Z","iopub.execute_input":"2022-08-03T03:29:33.447511Z","iopub.status.idle":"2022-08-03T03:29:33.504608Z","shell.execute_reply.started":"2022-08-03T03:29:33.447459Z","shell.execute_reply":"2022-08-03T03:29:33.503476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SVM Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\n\nSVClassifier = SVC()\n\nnp.average(cross_val_score(SVClassifier, x_scaled, y, cv = 5))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:29:35.286001Z","iopub.execute_input":"2022-08-03T03:29:35.286435Z","iopub.status.idle":"2022-08-03T03:29:35.406348Z","shell.execute_reply.started":"2022-08-03T03:29:35.286398Z","shell.execute_reply":"2022-08-03T03:29:35.405456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'C': np.arange(0.1, 50.1, 0.1), 'kernel' : ['linear', 'poly', 'rbf', 'sigmoid'],'degree': np.arange(1, 3, 1)}\n\nGSC_SVClassifier = GridSearchCV(estimator = SVClassifier, param_grid = params, cv =5)\n\nGSC_SVClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:29:37.343466Z","iopub.execute_input":"2022-08-03T03:29:37.343982Z","iopub.status.idle":"2022-08-03T03:37:43.371438Z","shell.execute_reply.started":"2022-08-03T03:29:37.343940Z","shell.execute_reply":"2022-08-03T03:37:43.370234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SVClassifier.set_params(**GSC_SVClassifier.best_params_)\n\nnp.average(cross_val_score(SVClassifier, x_scaled, y, cv = 5))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:42:33.108699Z","iopub.execute_input":"2022-08-03T03:42:33.110138Z","iopub.status.idle":"2022-08-03T03:42:33.227458Z","shell.execute_reply.started":"2022-08-03T03:42:33.110078Z","shell.execute_reply":"2022-08-03T03:42:33.226272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SVClassifier.fit(x_scaled, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:42:34.094819Z","iopub.execute_input":"2022-08-03T03:42:34.095290Z","iopub.status.idle":"2022-08-03T03:42:34.127684Z","shell.execute_reply.started":"2022-08-03T03:42:34.095254Z","shell.execute_reply":"2022-08-03T03:42:34.126382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overall Performance","metadata":{}},{"cell_type":"markdown","source":"#### Cross Validation Accuracy","metadata":{}},{"cell_type":"code","source":"classifiers = [LRClassifier, KNNClassifier,DTClassifier, RFClassifier, SVClassifier ]\n\ncv_scores = {}\n\nfor i in classifiers:\n    \n    cv_scores[i] = cross_val_score(i, x_scaled, y, cv = 10)\n    print(f'Classifier    : {i}')\n    print(f'mean accuracy : {np.average(cross_val_score(i, x_scaled, y, cv = 10))}')    \n    print(f'Daviation     : {np.std(cross_val_score(i, x_scaled, y, cv = 10))}\\n')\n    \n\nplt.figure(figsize = (10, 10))\nplt.title('Cross Validation Accuracy')\nsns.heatmap(pd.DataFrame(cv_scores), annot = True, cbar = False, cmap = 'Blues')\nplt.tick_params(rotation = 90, labelsize = 'large')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:42:42.594034Z","iopub.execute_input":"2022-08-03T03:42:42.594467Z","iopub.status.idle":"2022-08-03T03:42:46.551252Z","shell.execute_reply.started":"2022-08-03T03:42:42.594434Z","shell.execute_reply":"2022-08-03T03:42:46.550232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Precision and Recall","metadata":{}},{"cell_type":"code","source":"classifiers = [LRClassifier, KNNClassifier,DTClassifier, RFClassifier, SVClassifier ]\n\nfor i in classifiers:\n    \n    print(f'Classifier : {i}')\n    print(f'Precision  : {precision_score(y, i.predict(x_scaled))}')    \n    print(f'Recall     : {recall_score(y, i.predict(x_scaled))}\\n')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:43:05.475769Z","iopub.execute_input":"2022-08-03T03:43:05.476284Z","iopub.status.idle":"2022-08-03T03:43:05.710309Z","shell.execute_reply.started":"2022-08-03T03:43:05.476246Z","shell.execute_reply":"2022-08-03T03:43:05.708678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:44:10.439614Z","iopub.execute_input":"2022-08-03T03:44:10.440108Z","iopub.status.idle":"2022-08-03T03:44:10.463306Z","shell.execute_reply.started":"2022-08-03T03:44:10.440072Z","shell.execute_reply":"2022-08-03T03:44:10.462057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_transformed = df_transform(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:44:11.446349Z","iopub.execute_input":"2022-08-03T03:44:11.446842Z","iopub.status.idle":"2022-08-03T03:44:11.487574Z","shell.execute_reply.started":"2022-08-03T03:44:11.446803Z","shell.execute_reply":"2022-08-03T03:44:11.485767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['Survived'] = RFClassifier.predict(test_transformed)\n\nsubmission = test[['PassengerId', 'Survived']]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:44:12.386879Z","iopub.execute_input":"2022-08-03T03:44:12.387655Z","iopub.status.idle":"2022-08-03T03:44:12.404669Z","shell.execute_reply.started":"2022-08-03T03:44:12.387610Z","shell.execute_reply":"2022-08-03T03:44:12.403392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom IPython.display import HTML\n\nimport base64\n\n\ndef create_download_link(df, title = \"Download CSV file\", filename = \"Test_Submission.csv\"):  \n    csv = df.to_csv()\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = '<a download=\"{filename}\" href=\"data:text/csv;base64,{payload}\" target=\"_blank\">{title}</a>'\n    html = html.format(payload=payload,title=title,filename=filename)\n    return HTML(html)\n\n\ncreate_download_link(submission)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:44:15.733295Z","iopub.execute_input":"2022-08-03T03:44:15.733715Z","iopub.status.idle":"2022-08-03T03:44:15.749951Z","shell.execute_reply.started":"2022-08-03T03:44:15.733682Z","shell.execute_reply":"2022-08-03T03:44:15.748688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}