{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T08:21:47.128153Z","iopub.execute_input":"2022-08-01T08:21:47.128628Z","iopub.status.idle":"2022-08-01T08:21:47.140385Z","shell.execute_reply.started":"2022-08-01T08:21:47.128592Z","shell.execute_reply":"2022-08-01T08:21:47.139071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import warnings filter\nfrom warnings import simplefilter\n# ignore all future warnings\nsimplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:47.827761Z","iopub.execute_input":"2022-08-01T08:21:47.828763Z","iopub.status.idle":"2022-08-01T08:21:47.833707Z","shell.execute_reply.started":"2022-08-01T08:21:47.828729Z","shell.execute_reply":"2022-08-01T08:21:47.832488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nfrom matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:48.123831Z","iopub.execute_input":"2022-08-01T08:21:48.124256Z","iopub.status.idle":"2022-08-01T08:21:48.130574Z","shell.execute_reply.started":"2022-08-01T08:21:48.124214Z","shell.execute_reply":"2022-08-01T08:21:48.129004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:48.639631Z","iopub.execute_input":"2022-08-01T08:21:48.640327Z","iopub.status.idle":"2022-08-01T08:21:48.651270Z","shell.execute_reply.started":"2022-08-01T08:21:48.640281Z","shell.execute_reply":"2022-08-01T08:21:48.650204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:49.156064Z","iopub.execute_input":"2022-08-01T08:21:49.156765Z","iopub.status.idle":"2022-08-01T08:21:49.162955Z","shell.execute_reply.started":"2022-08-01T08:21:49.156730Z","shell.execute_reply":"2022-08-01T08:21:49.162086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df.isnull(), cbar = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:49.747585Z","iopub.execute_input":"2022-08-01T08:21:49.748206Z","iopub.status.idle":"2022-08-01T08:21:50.101164Z","shell.execute_reply.started":"2022-08-01T08:21:49.748173Z","shell.execute_reply":"2022-08-01T08:21:50.099830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:50.263714Z","iopub.execute_input":"2022-08-01T08:21:50.264205Z","iopub.status.idle":"2022-08-01T08:21:50.273960Z","shell.execute_reply.started":"2022-08-01T08:21:50.264169Z","shell.execute_reply":"2022-08-01T08:21:50.272920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x = df.columns, y = df.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:50.771801Z","iopub.execute_input":"2022-08-01T08:21:50.772958Z","iopub.status.idle":"2022-08-01T08:21:51.041627Z","shell.execute_reply.started":"2022-08-01T08:21:50.772916Z","shell.execute_reply":"2022-08-01T08:21:51.040265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PassangerID, Name, Ticket, Fare & Cabin can be dropped as :\n  \n- PassangerID and Name will not have any impact on survival \n\n- PClass already defines the class in which passanger is traveling. So, no need to consider ticket and fare\n\n- Cabin has around 80% null values\n    ","metadata":{}},{"cell_type":"code","source":"drop_cols = ['PassengerId', 'Name', 'Ticket', 'Fare', 'Cabin']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:51.363803Z","iopub.execute_input":"2022-08-01T08:21:51.364205Z","iopub.status.idle":"2022-08-01T08:21:51.369486Z","shell.execute_reply.started":"2022-08-01T08:21:51.364173Z","shell.execute_reply":"2022-08-01T08:21:51.368222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.loc[:, ~df.columns.isin(drop_cols)]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:51.879998Z","iopub.execute_input":"2022-08-01T08:21:51.880399Z","iopub.status.idle":"2022-08-01T08:21:51.887802Z","shell.execute_reply.started":"2022-08-01T08:21:51.880366Z","shell.execute_reply":"2022-08-01T08:21:51.886395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:52.399589Z","iopub.execute_input":"2022-08-01T08:21:52.400272Z","iopub.status.idle":"2022-08-01T08:21:52.416862Z","shell.execute_reply.started":"2022-08-01T08:21:52.400235Z","shell.execute_reply":"2022-08-01T08:21:52.415292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(df, hue = 'Survived')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:52.904114Z","iopub.execute_input":"2022-08-01T08:21:52.904536Z","iopub.status.idle":"2022-08-01T08:21:58.421935Z","shell.execute_reply.started":"2022-08-01T08:21:52.904503Z","shell.execute_reply":"2022-08-01T08:21:58.420904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical = ['Survived', 'Pclass', 'Age', 'SibSp', 'Parch']\n\ncategories = ['Sex', 'Embarked']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:58.424308Z","iopub.execute_input":"2022-08-01T08:21:58.424806Z","iopub.status.idle":"2022-08-01T08:21:58.430427Z","shell.execute_reply.started":"2022-08-01T08:21:58.424770Z","shell.execute_reply":"2022-08-01T08:21:58.429420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,3, figsize = (20,10))\n\nfor variable, subplot in zip(numerical, ax.flatten()):\n    \n    sns.histplot(df[variable],kde = True, ax = subplot)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:58.431712Z","iopub.execute_input":"2022-08-01T08:21:58.432679Z","iopub.status.idle":"2022-08-01T08:21:59.780764Z","shell.execute_reply.started":"2022-08-01T08:21:58.432647Z","shell.execute_reply":"2022-08-01T08:21:59.779420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize = (20, 10))\n\nfor variable, subplot in zip(categories, ax.flatten()):\n    \n    sns.countplot(x = variable, ax = subplot, hue = 'Survived', data = df)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:21:59.785006Z","iopub.execute_input":"2022-08-01T08:21:59.785543Z","iopub.status.idle":"2022-08-01T08:22:00.176928Z","shell.execute_reply.started":"2022-08-01T08:21:59.785493Z","shell.execute_reply":"2022-08-01T08:22:00.175735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,3, figsize = (30, 20))\n\nfor variable, subplots in zip(numerical, ax.flatten()):\n    \n    sns.histplot(x = variable, data = df, hue = 'Survived',kde = True, ax = subplots  )","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:00.178954Z","iopub.execute_input":"2022-08-01T08:22:00.179597Z","iopub.status.idle":"2022-08-01T08:22:02.169848Z","shell.execute_reply.started":"2022-08-01T08:22:00.179561Z","shell.execute_reply":"2022-08-01T08:22:02.168462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the plots above,\n\n- Pclass 3 has low rate of survival\n\n- age between 20 and 50 have more values and lesser survival rate than the other age groups\n\n- SibSp (Sibling and spouses) 0 and Parch (Parents and child) 0 have greater count but survival is low\n\n- SibSp, Parch and Age have some outliers","metadata":{}},{"cell_type":"markdown","source":"Outliers","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1,2, figsize = (20,10))\n\nplt.suptitle('SibSp Outlier Study')\n\nsns.countplot(df.SibSp, ax = axes[0])\nsns.boxplot(df.SibSp, ax = axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:02.171332Z","iopub.execute_input":"2022-08-01T08:22:02.171739Z","iopub.status.idle":"2022-08-01T08:22:02.560852Z","shell.execute_reply.started":"2022-08-01T08:22:02.171705Z","shell.execute_reply":"2022-08-01T08:22:02.559236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df['SibSp'] >4, :].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:02.563055Z","iopub.execute_input":"2022-08-01T08:22:02.563717Z","iopub.status.idle":"2022-08-01T08:22:02.576144Z","shell.execute_reply.started":"2022-08-01T08:22:02.563665Z","shell.execute_reply":"2022-08-01T08:22:02.573971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,2, figsize = (20,10))\n\nplt.suptitle('Parch Outlier Study')\n\nsns.countplot(df.Parch, ax = axes[0])\nsns.boxplot(df.Parch, ax = axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:02.577767Z","iopub.execute_input":"2022-08-01T08:22:02.578361Z","iopub.status.idle":"2022-08-01T08:22:02.943591Z","shell.execute_reply.started":"2022-08-01T08:22:02.578311Z","shell.execute_reply":"2022-08-01T08:22:02.942516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df['Parch'] >2, :].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:02.944969Z","iopub.execute_input":"2022-08-01T08:22:02.945416Z","iopub.status.idle":"2022-08-01T08:22:02.954278Z","shell.execute_reply.started":"2022-08-01T08:22:02.945382Z","shell.execute_reply":"2022-08-01T08:22:02.952909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1,2, figsize = (20,10))\n\nplt.suptitle('Age Outlier Study')\n\nsns.countplot(df.Age, ax = axes[0])\naxes[0].tick_params(rotation = 90, labelsize = 'large')\nsns.boxplot(df.Age, ax = axes[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:02.959525Z","iopub.execute_input":"2022-08-01T08:22:02.959939Z","iopub.status.idle":"2022-08-01T08:22:04.412545Z","shell.execute_reply.started":"2022-08-01T08:22:02.959905Z","shell.execute_reply":"2022-08-01T08:22:04.411084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[df['Age'] >65, :].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.414706Z","iopub.execute_input":"2022-08-01T08:22:04.415210Z","iopub.status.idle":"2022-08-01T08:22:04.424854Z","shell.execute_reply.started":"2022-08-01T08:22:04.415176Z","shell.execute_reply":"2022-08-01T08:22:04.423476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the analysis above, considering all the Parch >2, SibSp >3 and age > 65 as outliers.\n\nAs it comprises of around 6% of the total data, will delete those records.","metadata":{}},{"cell_type":"markdown","source":"# Data Cleanup and Preprocessing\n\n- We have two categorical Values (Sex, Embarked), we can use One hot Encoding for those values.\n\n- Need to take care the null values in age column, we will use KNN imputer for that.\n\n- There are Outliers (Age, Parch, SibSp) detected from the analysis.\n\n- Encode Age, Parch, SibSp based on the frequency of that data.\n\n- We are using Std scaler to scale the data.\n\n- Need to define a function to encode and prepare the test data as per the preprocessing.","metadata":{}},{"cell_type":"code","source":"df = pd.concat([df, pd.get_dummies(df['Embarked'])], axis = 1)\n\ndf.drop('Embarked', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.426621Z","iopub.execute_input":"2022-08-01T08:22:04.427085Z","iopub.status.idle":"2022-08-01T08:22:04.440610Z","shell.execute_reply.started":"2022-08-01T08:22:04.427036Z","shell.execute_reply":"2022-08-01T08:22:04.439143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df, pd.get_dummies(df['Sex'])], axis = 1)\n\ndf.drop('Sex', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.442128Z","iopub.execute_input":"2022-08-01T08:22:04.442738Z","iopub.status.idle":"2022-08-01T08:22:04.455156Z","shell.execute_reply.started":"2022-08-01T08:22:04.442701Z","shell.execute_reply":"2022-08-01T08:22:04.453930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df, pd.get_dummies(df['Pclass'])], axis = 1)\n\ndf.drop('Pclass', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.457275Z","iopub.execute_input":"2022-08-01T08:22:04.458237Z","iopub.status.idle":"2022-08-01T08:22:04.466881Z","shell.execute_reply.started":"2022-08-01T08:22:04.458201Z","shell.execute_reply":"2022-08-01T08:22:04.465848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.468261Z","iopub.execute_input":"2022-08-01T08:22:04.469194Z","iopub.status.idle":"2022-08-01T08:22:04.489730Z","shell.execute_reply.started":"2022-08-01T08:22:04.469161Z","shell.execute_reply":"2022-08-01T08:22:04.488785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(df['Age'], kde = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.491086Z","iopub.execute_input":"2022-08-01T08:22:04.491585Z","iopub.status.idle":"2022-08-01T08:22:04.783740Z","shell.execute_reply.started":"2022-08-01T08:22:04.491555Z","shell.execute_reply":"2022-08-01T08:22:04.782500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(df.loc[df.Age>65, :].index, inplace = True)\ndf.drop(df.loc[df.Parch>2].index, inplace = True)\ndf.drop(df.loc[df.SibSp>3].index, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.786050Z","iopub.execute_input":"2022-08-01T08:22:04.786552Z","iopub.status.idle":"2022-08-01T08:22:04.799166Z","shell.execute_reply.started":"2022-08-01T08:22:04.786500Z","shell.execute_reply":"2022-08-01T08:22:04.797960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df.loc[:, df.columns != 'Survived']\ny = df.loc[:, df.columns == 'Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.800811Z","iopub.execute_input":"2022-08-01T08:22:04.801406Z","iopub.status.idle":"2022-08-01T08:22:04.810274Z","shell.execute_reply.started":"2022-08-01T08:22:04.801336Z","shell.execute_reply":"2022-08-01T08:22:04.808841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors = 3)\n\nx_trans = pd.DataFrame(imputer.fit_transform(x), columns = x.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.811721Z","iopub.execute_input":"2022-08-01T08:22:04.812073Z","iopub.status.idle":"2022-08-01T08:22:04.842353Z","shell.execute_reply.started":"2022-08-01T08:22:04.812010Z","shell.execute_reply":"2022-08-01T08:22:04.840727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(x_trans['Age'], kde = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:04.844325Z","iopub.execute_input":"2022-08-01T08:22:04.845451Z","iopub.status.idle":"2022-08-01T08:22:05.176123Z","shell.execute_reply.started":"2022-08-01T08:22:04.845392Z","shell.execute_reply":"2022-08-01T08:22:05.175075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_trans.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:05.177572Z","iopub.execute_input":"2022-08-01T08:22:05.178473Z","iopub.status.idle":"2022-08-01T08:22:05.189498Z","shell.execute_reply.started":"2022-08-01T08:22:05.178436Z","shell.execute_reply":"2022-08-01T08:22:05.188175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Null Values are Handled using KNN Imputer without significant change in distribution","metadata":{}},{"cell_type":"markdown","source":"Handling Outliers\n\nParch >2, SibSp >3 and age > 65","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Scaling the Data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:05.196824Z","iopub.execute_input":"2022-08-01T08:22:05.197185Z","iopub.status.idle":"2022-08-01T08:22:05.203054Z","shell.execute_reply.started":"2022-08-01T08:22:05.197152Z","shell.execute_reply":"2022-08-01T08:22:05.201512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_scaled = pd.DataFrame(scaler.fit_transform(x_trans), columns = x_trans.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:05.204803Z","iopub.execute_input":"2022-08-01T08:22:05.205316Z","iopub.status.idle":"2022-08-01T08:22:05.218788Z","shell.execute_reply.started":"2022-08-01T08:22:05.205253Z","shell.execute_reply":"2022-08-01T08:22:05.217758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize = (20, 20))\n\n\nfor variable, subplots in zip(x_scaled.columns, axes.flatten()):\n    \n    sns.histplot(x_scaled[variable], kde = True, ax = subplots)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:05.220244Z","iopub.execute_input":"2022-08-01T08:22:05.221470Z","iopub.status.idle":"2022-08-01T08:22:07.462989Z","shell.execute_reply.started":"2022-08-01T08:22:05.221422Z","shell.execute_reply":"2022-08-01T08:22:07.461642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Defining a transformation Function\n\nOperations :\n\n\n- Drop PassengerID, Name, Ticket, Fare, cabin\n\n- Imputer to handle null values\n\n- One hot encoding for Pclass, Sex, Embarked\n\n- MinMax Scaling for the rest","metadata":{}},{"cell_type":"code","source":"def df_transform(df):\n    \n    # Dropping columns which are not considered\n    \n    drop_cols = ['PassengerId', 'Name', 'Ticket', 'Fare', 'Cabin']    \n    df = df.loc[:, ~df.columns.isin(drop_cols)]\n    \n    # One hot encoding for Embarked\n    \n    df = pd.concat([df, pd.get_dummies(df['Embarked'])], axis = 1)\n    df.drop('Embarked', axis = 1, inplace = True)\n    \n    # One hot encoding for sex\n    \n    df = pd.concat([df, pd.get_dummies(df['Sex'])], axis = 1)\n    df.drop('Sex', axis = 1, inplace = True)\n    \n    # One hot encoding for Pclass\n    \n    df = pd.concat([df, pd.get_dummies(df['Pclass'])], axis = 1)\n    df.drop('Pclass', axis = 1, inplace = True)\n    \n    # Null Handling using inputer\n    \n    df = pd.DataFrame(imputer.transform(df), columns = df.columns)\n    \n    #Scaling the data\n    \n    df = pd.DataFrame(scaler.transform(df), columns = df.columns)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:07.464987Z","iopub.execute_input":"2022-08-01T08:22:07.465777Z","iopub.status.idle":"2022-08-01T08:22:07.478142Z","shell.execute_reply.started":"2022-08-01T08:22:07.465726Z","shell.execute_reply":"2022-08-01T08:22:07.477137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training and Validation","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:07.480636Z","iopub.execute_input":"2022-08-01T08:22:07.480950Z","iopub.status.idle":"2022-08-01T08:22:07.491212Z","shell.execute_reply.started":"2022-08-01T08:22:07.480919Z","shell.execute_reply":"2022-08-01T08:22:07.490116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(x_scaled, y, train_size = 0.85)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:07.492709Z","iopub.execute_input":"2022-08-01T08:22:07.493104Z","iopub.status.idle":"2022-08-01T08:22:07.504376Z","shell.execute_reply.started":"2022-08-01T08:22:07.493061Z","shell.execute_reply":"2022-08-01T08:22:07.503442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:07.505898Z","iopub.execute_input":"2022-08-01T08:22:07.507215Z","iopub.status.idle":"2022-08-01T08:22:07.516990Z","shell.execute_reply.started":"2022-08-01T08:22:07.507177Z","shell.execute_reply":"2022-08-01T08:22:07.516004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nlogistic_regression = LogisticRegression()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:07.518397Z","iopub.execute_input":"2022-08-01T08:22:07.519497Z","iopub.status.idle":"2022-08-01T08:22:07.527861Z","shell.execute_reply.started":"2022-08-01T08:22:07.519450Z","shell.execute_reply":"2022-08-01T08:22:07.526495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_regression.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:07.731391Z","iopub.execute_input":"2022-08-01T08:22:07.732390Z","iopub.status.idle":"2022-08-01T08:22:07.758228Z","shell.execute_reply.started":"2022-08-01T08:22:07.732355Z","shell.execute_reply":"2022-08-01T08:22:07.757205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_regression.score(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:08.771915Z","iopub.execute_input":"2022-08-01T08:22:08.772655Z","iopub.status.idle":"2022-08-01T08:22:08.783953Z","shell.execute_reply.started":"2022-08-01T08:22:08.772602Z","shell.execute_reply":"2022-08-01T08:22:08.782599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_regression.score(x_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:09.475856Z","iopub.execute_input":"2022-08-01T08:22:09.476321Z","iopub.status.idle":"2022-08-01T08:22:09.488671Z","shell.execute_reply.started":"2022-08-01T08:22:09.476286Z","shell.execute_reply":"2022-08-01T08:22:09.487181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(confusion_matrix(y_test, logistic_regression.predict(x_test)), annot = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:10.343862Z","iopub.execute_input":"2022-08-01T08:22:10.344288Z","iopub.status.idle":"2022-08-01T08:22:10.600121Z","shell.execute_reply.started":"2022-08-01T08:22:10.344255Z","shell.execute_reply":"2022-08-01T08:22:10.599175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\nTest')\nprint(classification_report(y_test, logistic_regression.predict(x_test)))\nprint('\\nTrain')\nprint(classification_report(y_train, logistic_regression.predict(x_train)))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:11.183878Z","iopub.execute_input":"2022-08-01T08:22:11.184348Z","iopub.status.idle":"2022-08-01T08:22:11.209783Z","shell.execute_reply.started":"2022-08-01T08:22:11.184308Z","shell.execute_reply":"2022-08-01T08:22:11.208404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## KNN Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nKNNClassifier = KNeighborsClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:11.715522Z","iopub.execute_input":"2022-08-01T08:22:11.715919Z","iopub.status.idle":"2022-08-01T08:22:11.721925Z","shell.execute_reply.started":"2022-08-01T08:22:11.715887Z","shell.execute_reply":"2022-08-01T08:22:11.720526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KNNClassifier.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:12.257254Z","iopub.execute_input":"2022-08-01T08:22:12.258105Z","iopub.status.idle":"2022-08-01T08:22:12.272841Z","shell.execute_reply.started":"2022-08-01T08:22:12.258067Z","shell.execute_reply":"2022-08-01T08:22:12.271429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KNNClassifier.score(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:12.979578Z","iopub.execute_input":"2022-08-01T08:22:12.980387Z","iopub.status.idle":"2022-08-01T08:22:13.027621Z","shell.execute_reply.started":"2022-08-01T08:22:12.980347Z","shell.execute_reply":"2022-08-01T08:22:13.026429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KNNClassifier.score(x_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:14.071640Z","iopub.execute_input":"2022-08-01T08:22:14.072088Z","iopub.status.idle":"2022-08-01T08:22:14.090179Z","shell.execute_reply.started":"2022-08-01T08:22:14.072044Z","shell.execute_reply":"2022-08-01T08:22:14.088888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KNN_Train_Acc = {}\nKNN_Test_Acc = {}\n\nfor i in range(1, 50):\n    \n    KNNClassifier = KNeighborsClassifier(i)\n    \n    KNNClassifier.fit(x_train, y_train)\n    \n    KNN_Train_Acc[i] = KNNClassifier.score(x_train, y_train)\n    \n    KNN_Test_Acc[i] = KNNClassifier.score(x_test, y_test)\n    \n    print('\\ni : ', i)\n    print('Train Accuracy : ', KNNClassifier.score(x_train, y_train))\n    print('Test Accuracy : ', KNNClassifier.score(x_test, y_test))\n    \n\nplt.plot(KNN_Train_Acc.keys(), KNN_Train_Acc.values(), color = 'green')\nplt.plot(KNN_Test_Acc.keys(), KNN_Test_Acc.values(), color = 'red')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:14.655713Z","iopub.execute_input":"2022-08-01T08:22:14.656562Z","iopub.status.idle":"2022-08-01T08:22:20.119155Z","shell.execute_reply.started":"2022-08-01T08:22:14.656512Z","shell.execute_reply":"2022-08-01T08:22:20.117687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above analysis, it is seen that after i = 5, train and test accuracy are maximum","metadata":{}},{"cell_type":"code","source":"KNNClassifier = KNeighborsClassifier(5)\n\nKNNClassifier.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:23:59.503546Z","iopub.execute_input":"2022-08-01T08:23:59.503985Z","iopub.status.idle":"2022-08-01T08:23:59.519308Z","shell.execute_reply.started":"2022-08-01T08:23:59.503950Z","shell.execute_reply":"2022-08-01T08:23:59.518116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(confusion_matrix(y_test, KNNClassifier.predict(x_test)), annot = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:24:03.658751Z","iopub.execute_input":"2022-08-01T08:24:03.659194Z","iopub.status.idle":"2022-08-01T08:24:03.936255Z","shell.execute_reply.started":"2022-08-01T08:24:03.659160Z","shell.execute_reply":"2022-08-01T08:24:03.935048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\nClassification report : Train')\nprint(classification_report(y_train, KNNClassifier.predict(x_train)))\n\nprint('\\nClassification report : Test')\nprint(classification_report(y_test, KNNClassifier.predict(x_test)))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:24:11.934416Z","iopub.execute_input":"2022-08-01T08:24:11.934823Z","iopub.status.idle":"2022-08-01T08:24:11.998632Z","shell.execute_reply.started":"2022-08-01T08:24:11.934790Z","shell.execute_reply":"2022-08-01T08:24:11.997402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Decision Tree","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\n\nfrom sklearn import tree","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:20.449041Z","iopub.execute_input":"2022-08-01T08:22:20.449630Z","iopub.status.idle":"2022-08-01T08:22:20.454729Z","shell.execute_reply.started":"2022-08-01T08:22:20.449597Z","shell.execute_reply":"2022-08-01T08:22:20.453623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decisiontree = DecisionTreeClassifier()\ndecisiontree.fit(x_train, y_train)\nprint('\\nTrain Accuracy Score')\nprint(decisiontree.score(x_train, y_train))\nprint('\\nTest Accuracy Score')\nprint(decisiontree.score(x_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:20.456005Z","iopub.execute_input":"2022-08-01T08:22:20.456362Z","iopub.status.idle":"2022-08-01T08:22:20.476093Z","shell.execute_reply.started":"2022-08-01T08:22:20.456331Z","shell.execute_reply":"2022-08-01T08:22:20.474732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Clearly, decision tree is overfitted\n\nWe need to do some pruning","metadata":{}},{"cell_type":"code","source":"decisiontree.get_depth()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:20.477628Z","iopub.execute_input":"2022-08-01T08:22:20.478170Z","iopub.status.idle":"2022-08-01T08:22:20.484559Z","shell.execute_reply.started":"2022-08-01T08:22:20.478137Z","shell.execute_reply":"2022-08-01T08:22:20.483659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Tree_Train_Acc = {}\nTree_Test_Acc = {}\n\nfor i in range(2, decisiontree.get_depth()) :\n    \n    decisiontree = DecisionTreeClassifier(max_depth = i)\n    decisiontree.fit(x_train, y_train)\n    \n    print('\\nmax_depth : ', i)\n    print('Train Accuracy Score : ', decisiontree.score(x_train, y_train))\n    print('Test Accuracy Score : ', decisiontree.score(x_test, y_test))\n    \n    Tree_Train_Acc[i] = decisiontree.score(x_train, y_train)\n    Tree_Test_Acc[i] = decisiontree.score(x_test, y_test)\n    \n    \nplt.plot(Tree_Train_Acc.keys(), Tree_Train_Acc.values(), color = 'green')\nplt.plot(Tree_Test_Acc.keys(), Tree_Test_Acc.values(), color = 'red')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:20.485851Z","iopub.execute_input":"2022-08-01T08:22:20.486881Z","iopub.status.idle":"2022-08-01T08:22:20.837716Z","shell.execute_reply.started":"2022-08-01T08:22:20.486827Z","shell.execute_reply":"2022-08-01T08:22:20.836415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ccp_alpha tuning","metadata":{}},{"cell_type":"code","source":"ccp_alpha = np.linspace(0.01, 0.11, 10)\n\nTree_Train_Acc = {}\nTree_Test_Acc = {}\n\nfor i in ccp_alpha :\n    \n    decisiontree = DecisionTreeClassifier(ccp_alpha = i)\n    decisiontree.fit(x_train, y_train)\n    \n    print('\\nccp_alpha : ', i)\n    print('Train Accuracy Score : ', decisiontree.score(x_train, y_train))\n    print('Test Accuracy Score : ', decisiontree.score(x_test, y_test))\n    \n    Tree_Train_Acc[i] = decisiontree.score(x_train, y_train)\n    Tree_Test_Acc[i] = decisiontree.score(x_test, y_test)\n    \n    \nplt.plot(Tree_Train_Acc.keys(), Tree_Train_Acc.values(), color = 'green')\nplt.plot(Tree_Test_Acc.keys(), Tree_Test_Acc.values(), color = 'red')\n\nplt.show()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:24:50.759202Z","iopub.execute_input":"2022-08-01T08:24:50.759616Z","iopub.status.idle":"2022-08-01T08:24:51.073846Z","shell.execute_reply.started":"2022-08-01T08:24:50.759584Z","shell.execute_reply":"2022-08-01T08:24:51.072639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From above analysis,\n\nTree with depth 5 shows good correlation between bias and variance \n","metadata":{}},{"cell_type":"code","source":"decisiontree = DecisionTreeClassifier(max_depth = 5)\ndecisiontree.fit(x_train, y_train)\n\nprint('Train Accuracy Score : ', decisiontree.score(x_train, y_train))\nprint('Test Accuracy Score : ', decisiontree.score(x_test, y_test))\n\nprint('\\nClassification Report : ')\n\nprint(classification_report(y_test, decisiontree.predict(x_test)))\n\n\nsns.heatmap(confusion_matrix(y_test, decisiontree.predict(x_test)), annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:25:07.478320Z","iopub.execute_input":"2022-08-01T08:25:07.478726Z","iopub.status.idle":"2022-08-01T08:25:07.762778Z","shell.execute_reply.started":"2022-08-01T08:25:07.478673Z","shell.execute_reply":"2022-08-01T08:25:07.761326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Forrest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nRandomForest = RandomForestClassifier(n_estimators = 10, criterion = 'entropy')\n\nRandomForest.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:21.446322Z","iopub.execute_input":"2022-08-01T08:22:21.446656Z","iopub.status.idle":"2022-08-01T08:22:21.483275Z","shell.execute_reply.started":"2022-08-01T08:22:21.446608Z","shell.execute_reply":"2022-08-01T08:22:21.482304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RandomForest.score(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:21.484564Z","iopub.execute_input":"2022-08-01T08:22:21.485092Z","iopub.status.idle":"2022-08-01T08:22:21.497762Z","shell.execute_reply.started":"2022-08-01T08:22:21.485052Z","shell.execute_reply":"2022-08-01T08:22:21.496504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RandomForest.score(x_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:21.499521Z","iopub.execute_input":"2022-08-01T08:22:21.499861Z","iopub.status.idle":"2022-08-01T08:22:21.513964Z","shell.execute_reply.started":"2022-08-01T08:22:21.499827Z","shell.execute_reply":"2022-08-01T08:22:21.512484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Tree_Train_Acc = {}\nTree_Test_Acc = {}\n\nfor i in range(2, 20):\n    \n    RandomForest = RandomForestClassifier(max_depth = i)\n\n    RandomForest.fit(x_train, y_train) \n    print('\\nMax Depth : ', i)\n    print('Train Accuracy Score : ', RandomForest.score(x_train, y_train))\n    print('Test Accuracy Score : ', RandomForest.score(x_test, y_test))\n    \n    Tree_Train_Acc[i] = RandomForest.score(x_train, y_train)\n    Tree_Test_Acc[i] = RandomForest.score(x_test, y_test)   \n    \nplt.plot(Tree_Train_Acc.keys(), Tree_Train_Acc.values(), color = 'green')\nplt.plot(Tree_Test_Acc.keys(), Tree_Test_Acc.values(), color = 'red')\n\nplt.show()\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:21.517145Z","iopub.execute_input":"2022-08-01T08:22:21.517542Z","iopub.status.idle":"2022-08-01T08:22:27.340825Z","shell.execute_reply.started":"2022-08-01T08:22:21.517510Z","shell.execute_reply":"2022-08-01T08:22:27.339613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tree with max depth 5 shows good bias variance tradeoff","metadata":{}},{"cell_type":"code","source":"Tree_Train_Acc = {}\nTree_Test_Acc = {}\n\nfor i in range(2, 100):\n    \n    RandomForest = RandomForestClassifier(max_depth = 5, n_estimators = i)\n\n    RandomForest.fit(x_train, y_train) \n    print('\\nNo of Trees : ', i)\n    print('Train Accuracy Score : ', RandomForest.score(x_train, y_train))\n    print('Test Accuracy Score : ', RandomForest.score(x_test, y_test))\n    \n    Tree_Train_Acc[i] = RandomForest.score(x_train, y_train)\n    Tree_Test_Acc[i] = RandomForest.score(x_test, y_test)   \n    \nplt.plot(Tree_Train_Acc.keys(), Tree_Train_Acc.values(), color = 'green')\nplt.plot(Tree_Test_Acc.keys(), Tree_Test_Acc.values(), color = 'red')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:25:43.504716Z","iopub.execute_input":"2022-08-01T08:25:43.505234Z","iopub.status.idle":"2022-08-01T08:25:59.237434Z","shell.execute_reply.started":"2022-08-01T08:25:43.505197Z","shell.execute_reply":"2022-08-01T08:25:59.236088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    RandomForest = RandomForestClassifier(max_depth = 4)\n\n    RandomForest.fit(x_train, y_train) \n    print('\\n')\n    print('Train Accuracy Score : ', RandomForest.score(x_train, y_train))\n    print('Test Accuracy Score : ', RandomForest.score(x_test, y_test))\n    \n    print(classification_report(y_test, RandomForest.predict(x_test)))\n    \n    sns.heatmap(confusion_matrix(y_test, RandomForest.predict(x_test)), annot = True)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:27:07.869940Z","iopub.execute_input":"2022-08-01T08:27:07.870384Z","iopub.status.idle":"2022-08-01T08:27:08.442228Z","shell.execute_reply.started":"2022-08-01T08:27:07.870350Z","shell.execute_reply":"2022-08-01T08:27:08.440850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Support Vector Machines","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\n\nSVClassifier = SVC()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:43.026547Z","iopub.execute_input":"2022-08-01T08:22:43.026941Z","iopub.status.idle":"2022-08-01T08:22:43.032591Z","shell.execute_reply.started":"2022-08-01T08:22:43.026908Z","shell.execute_reply":"2022-08-01T08:22:43.031295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kernel = ['linear', 'poly', 'rbf', 'sigmoid']\n\nSVC_Train_Acc = {}\nSVC_Test_Acc = {}\n\nfor i in kernel:\n    \n    SVClassifier = SVC(kernel = i)\n    \n    SVClassifier.fit(x_train, y_train)\n    \n    SVC_Train_Acc[i] = SVClassifier.score(x_train, y_train)\n    SVC_Test_Acc[i] = SVClassifier.score(x_test, y_test)\n    \n    print('\\nKernel : ', i)\n    print('SVC Train Acc : ', SVClassifier.score(x_train, y_train))\n    print('SVC Test Acc : ', SVClassifier.score(x_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:22:43.033950Z","iopub.execute_input":"2022-08-01T08:22:43.034334Z","iopub.status.idle":"2022-08-01T08:22:43.255134Z","shell.execute_reply.started":"2022-08-01T08:22:43.034302Z","shell.execute_reply":"2022-08-01T08:22:43.253731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c = np.linspace(0.01, 10.01, 100 )\n\nSVC_Train_Acc = {}\nSVC_Test_Acc = {}\n\nfor i in c:\n    \n    SVClassifier = SVC(kernel = 'poly', C = i)\n    \n    SVClassifier.fit(x_train, y_train)\n    \n    SVC_Train_Acc[i] = SVClassifier.score(x_train, y_train)\n    SVC_Test_Acc[i] = SVClassifier.score(x_test, y_test)\n    \n    print('\\nC : ', i)\n    print('SVC Train Acc : ', SVClassifier.score(x_train, y_train))\n    print('SVC Test Acc : ', SVClassifier.score(x_test, y_test))\n    \nplt.plot(SVC_Train_Acc.keys(), SVC_Train_Acc.values(), color = 'green')\nplt.plot(SVC_Test_Acc.keys(), SVC_Test_Acc.values(), color = 'red')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:29:20.025070Z","iopub.execute_input":"2022-08-01T08:29:20.025504Z","iopub.status.idle":"2022-08-01T08:29:26.024220Z","shell.execute_reply.started":"2022-08-01T08:29:20.025470Z","shell.execute_reply":"2022-08-01T08:29:26.022865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"around c = 4.5, there is good correlation between test and train accuracy","metadata":{}},{"cell_type":"code","source":"SVClassifier = SVC(kernel = 'poly', C = 4.5)\n    \nSVClassifier.fit(x_train, y_train)\n\nprint('\\n')\nprint('SVC Train Acc : ', SVClassifier.score(x_train, y_train))\nprint('SVC Test Acc : ', SVClassifier.score(x_test, y_test))\n\nprint('\\nClassification Report : ')\n\nprint(classification_report(y_test, SVClassifier.predict(x_test)))\n\nsns.heatmap(confusion_matrix(y_test, SVClassifier.predict(x_test)), annot = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:30:07.488818Z","iopub.execute_input":"2022-08-01T08:30:07.489257Z","iopub.status.idle":"2022-08-01T08:30:07.812909Z","shell.execute_reply.started":"2022-08-01T08:30:07.489222Z","shell.execute_reply":"2022-08-01T08:30:07.811577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifiers = [logistic_regression, KNNClassifier, decisiontree, RandomForest, SVClassifier]\n#classifiers = [decisiontree,SVClassifier]\n\nfor i in classifiers:\n    \n    print('\\nClassifier : ', i)\n    \n    print('Train Acc : ', i.score(x_train, y_train))\n    print('Test Acc : ', i.score(x_test, y_test))\n    print('\\n')\n    print(classification_report(y_test, i.predict(x_test)))\n \n    sns.heatmap(confusion_matrix(y_test, i.predict(x_test)), annot = True, cbar = False)\n    \n    plt.show()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:30:16.409624Z","iopub.execute_input":"2022-08-01T08:30:16.410091Z","iopub.status.idle":"2022-08-01T08:30:17.352845Z","shell.execute_reply.started":"2022-08-01T08:30:16.410056Z","shell.execute_reply":"2022-08-01T08:30:17.351408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From analysis above, KNN is performing well overall","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:32:21.143519Z","iopub.execute_input":"2022-08-01T08:32:21.144003Z","iopub.status.idle":"2022-08-01T08:32:21.159497Z","shell.execute_reply.started":"2022-08-01T08:32:21.143968Z","shell.execute_reply":"2022-08-01T08:32:21.157936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_transformed = df_transform(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:33:25.749700Z","iopub.execute_input":"2022-08-01T08:33:25.750155Z","iopub.status.idle":"2022-08-01T08:33:25.784923Z","shell.execute_reply.started":"2022-08-01T08:33:25.750122Z","shell.execute_reply":"2022-08-01T08:33:25.783184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['Survived'] = KNNClassifier.predict(test_transformed)\n\nsubmission = test[['PassengerId', 'Survived']]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:51:15.548864Z","iopub.execute_input":"2022-08-01T08:51:15.549281Z","iopub.status.idle":"2022-08-01T08:51:15.581545Z","shell.execute_reply.started":"2022-08-01T08:51:15.549248Z","shell.execute_reply":"2022-08-01T08:51:15.580590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\n\nimport base64\n\n\ndef create_download_link(df, title = \"Download CSV file\", filename = \"data.csv\"):  \n    csv = df.to_csv()\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = '<a download=\"{filename}\" href=\"data:text/csv;base64,{payload}\" target=\"_blank\">{title}</a>'\n    html = html.format(payload=payload,title=title,filename=filename)\n    return HTML(html)\n\n\ncreate_download_link(submission)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:51:28.070708Z","iopub.execute_input":"2022-08-01T08:51:28.071148Z","iopub.status.idle":"2022-08-01T08:51:28.083719Z","shell.execute_reply.started":"2022-08-01T08:51:28.071115Z","shell.execute_reply":"2022-08-01T08:51:28.082761Z"},"trusted":true},"execution_count":null,"outputs":[]}]}