{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## Import basic libraries\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T11:08:53.554324Z","iopub.execute_input":"2022-07-22T11:08:53.554766Z","iopub.status.idle":"2022-07-22T11:08:53.560489Z","shell.execute_reply.started":"2022-07-22T11:08:53.554717Z","shell.execute_reply":"2022-07-22T11:08:53.559276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/titanic/train.csv')\ndata.head()","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2022-07-22T11:08:53.567789Z","iopub.execute_input":"2022-07-22T11:08:53.568448Z","iopub.status.idle":"2022-07-22T11:08:53.604108Z","shell.execute_reply.started":"2022-07-22T11:08:53.568398Z","shell.execute_reply":"2022-07-22T11:08:53.603273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.605763Z","iopub.execute_input":"2022-07-22T11:08:53.606246Z","iopub.status.idle":"2022-07-22T11:08:53.620978Z","shell.execute_reply.started":"2022-07-22T11:08:53.606213Z","shell.execute_reply":"2022-07-22T11:08:53.619810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.622604Z","iopub.execute_input":"2022-07-22T11:08:53.623439Z","iopub.status.idle":"2022-07-22T11:08:53.661780Z","shell.execute_reply.started":"2022-07-22T11:08:53.623390Z","shell.execute_reply":"2022-07-22T11:08:53.660542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove unwanted attributes\ndata = data.drop(['PassengerId','Ticket'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.663312Z","iopub.execute_input":"2022-07-22T11:08:53.663985Z","iopub.status.idle":"2022-07-22T11:08:53.671699Z","shell.execute_reply.started":"2022-07-22T11:08:53.663937Z","shell.execute_reply":"2022-07-22T11:08:53.670574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Extraction","metadata":{}},{"cell_type":"markdown","source":"## Cabin","metadata":{}},{"cell_type":"code","source":"data.Cabin.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.673596Z","iopub.execute_input":"2022-07-22T11:08:53.674122Z","iopub.status.idle":"2022-07-22T11:08:53.685597Z","shell.execute_reply.started":"2022-07-22T11:08:53.674082Z","shell.execute_reply":"2022-07-22T11:08:53.684759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_decks(df):\n    temp = []\n    for i in df['Cabin'].unique():\n        a = str(i)[0]\n        if a not in temp:\n            temp.append(a)\n    return temp\n\nshow_decks(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.687233Z","iopub.execute_input":"2022-07-22T11:08:53.687739Z","iopub.status.idle":"2022-07-22T11:08:53.698149Z","shell.execute_reply.started":"2022-07-22T11:08:53.687699Z","shell.execute_reply":"2022-07-22T11:08:53.696563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](http://upload.wikimedia.org/wikipedia/commons/0/0d/Olympic_%26_Titanic_cutaway_diagram.png)","metadata":{}},{"cell_type":"markdown","source":"### Implementing Cabin Transformer\n* Extract the first alphabetic character of the Cabin which is a deck of the ship\n* Group the decks level wise from top to bottom","metadata":{}},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\n\nclass CabinTransformer(BaseEstimator, TransformerMixin):\n    def fit(self, df):\n        return self\n    \n    def extract_cabin(self, cabin):\n        cabin = str(cabin)\n        if cabin=='nan':\n            return 'U'\n        else:\n            return cabin[0]\n\n    def group_deck(self, deck):\n        if deck in ['A','B','C','T']:\n            return 'ABC'  # Top 3 Decks\n        elif deck in ['D','E']:\n            return 'DE'  # Next 2 Decks from top\n        elif deck in ['F','G']:\n            return 'FG'  # Bottom 2 Decks\n        else:\n            return 'U'\n        \n    def transform(self, df):\n        df1 = df.copy()\n        df1['Cabin'] = df1['Cabin'].apply(self.extract_cabin)\n        df1['Cabin'] = df1['Cabin'].apply(self.group_deck)\n        \n        cabin_maps = {'ABC':0, 'DE':1, 'FG':2, 'U':3}\n        df1['Cabin'] = df1['Cabin'].map(cabin_maps)\n        return df1","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.699469Z","iopub.execute_input":"2022-07-22T11:08:53.699798Z","iopub.status.idle":"2022-07-22T11:08:53.714595Z","shell.execute_reply.started":"2022-07-22T11:08:53.699765Z","shell.execute_reply":"2022-07-22T11:08:53.713110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CabinTransformer().fit_transform(data).head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.716535Z","iopub.execute_input":"2022-07-22T11:08:53.717296Z","iopub.status.idle":"2022-07-22T11:08:53.748567Z","shell.execute_reply.started":"2022-07-22T11:08:53.717243Z","shell.execute_reply":"2022-07-22T11:08:53.747538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Cabin',hue = 'Survived', data=CabinTransformer().fit_transform(data))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.750423Z","iopub.execute_input":"2022-07-22T11:08:53.750744Z","iopub.status.idle":"2022-07-22T11:08:53.926525Z","shell.execute_reply.started":"2022-07-22T11:08:53.750713Z","shell.execute_reply":"2022-07-22T11:08:53.925618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Name\nName of the passenger contains his/her title and Rank on the ship will be a usefull information","metadata":{}},{"cell_type":"code","source":"temp = []\nfor name in data['Name']:\n    a = name.split(', ')[1].split('.')[0]\n    if a not in temp:\n        temp.append(a)\nprint('Titles present on the ship:\\n', temp)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.928524Z","iopub.execute_input":"2022-07-22T11:08:53.929172Z","iopub.status.idle":"2022-07-22T11:08:53.939616Z","shell.execute_reply.started":"2022-07-22T11:08:53.929123Z","shell.execute_reply":"2022-07-22T11:08:53.938533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Implementing Name Column Transformer\n* Extract titles from the name of the passenger\n* Group the titles according to Rank they hold","metadata":{}},{"cell_type":"code","source":"class CustomAttributeTitle(BaseEstimator, TransformerMixin):\n    def fit(self, df):\n        return self\n    \n    def title(self, name):\n        return name.split(', ')[1].split('.')[0]\n    \n    def group(self, title):\n        if title in ['Miss', 'Mlle', 'Ms']:\n            return 'Miss'\n        \n        if title in ['Mrs', 'Mme']:\n            return 'Mrs'\n        \n        if title in ['Lady', 'the Countess','Capt', 'Col','Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona']:\n            return 'Rare'\n        \n        else:\n            return title\n        \n    \n    def transform(self, df):\n        df1 = df.copy()\n        df1['Title'] = df1['Name'].apply(self.title)\n        df1['Title'] = df1['Title'].apply(self.group)\n        \n        title_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 0}\n        df1['Title'] = df1['Title'].map(title_mapping)\n        df1['Title'] = df1['Title'].fillna(0)\n        df1 = df1.drop('Name', axis=1)\n        \n        return df1","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.941417Z","iopub.execute_input":"2022-07-22T11:08:53.941799Z","iopub.status.idle":"2022-07-22T11:08:53.955869Z","shell.execute_reply.started":"2022-07-22T11:08:53.941762Z","shell.execute_reply":"2022-07-22T11:08:53.954873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CustomAttributeTitle().fit_transform(data).head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:53.957404Z","iopub.execute_input":"2022-07-22T11:08:53.957898Z","iopub.status.idle":"2022-07-22T11:08:54.001478Z","shell.execute_reply.started":"2022-07-22T11:08:53.957852Z","shell.execute_reply":"2022-07-22T11:08:54.000429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#f = plt.figure(figsize=(12,8))\nsns.countplot(x='Title',hue = 'Survived', data=CustomAttributeTitle().fit_transform(data))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.002939Z","iopub.execute_input":"2022-07-22T11:08:54.003248Z","iopub.status.idle":"2022-07-22T11:08:54.191351Z","shell.execute_reply.started":"2022-07-22T11:08:54.003219Z","shell.execute_reply":"2022-07-22T11:08:54.190373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Age","metadata":{}},{"cell_type":"code","source":"plt.hist(data['Age'], bins=5)[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.195051Z","iopub.execute_input":"2022-07-22T11:08:54.195361Z","iopub.status.idle":"2022-07-22T11:08:54.375711Z","shell.execute_reply.started":"2022-07-22T11:08:54.195330Z","shell.execute_reply":"2022-07-22T11:08:54.374599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def AgeBand(df):\n    df1 = df.copy()\n    df1['AgeBand'] = pd.cut(df1['Age'], 5)\n    return df1[['AgeBand', 'Survived']].groupby(['AgeBand'], as_index=False).mean().sort_values(by='AgeBand', ascending=True)\n\nAgeBand(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.379826Z","iopub.execute_input":"2022-07-22T11:08:54.380160Z","iopub.status.idle":"2022-07-22T11:08:54.406352Z","shell.execute_reply.started":"2022-07-22T11:08:54.380129Z","shell.execute_reply":"2022-07-22T11:08:54.405307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Implementing Age Column transformer\n* Impute the missing values in the age column by using the median values of each title. As a result no master will be in the age of 30.","metadata":{}},{"cell_type":"code","source":"class AgeTransformer(BaseEstimator, TransformerMixin):\n    def fit(self, df):\n        self.means = {}\n        for i in df['Title'].unique():\n            m = df[df['Title']==i]['Age'].median()\n            self.means[i] = m\n        return self\n    \n    def transform(self, df):\n        df1 = df.copy()\n        index_values = df1.index.values.astype(int)\n        for i in index_values:\n            age = df1.at[i, 'Age'].astype(float)\n            if np.isnan(age):\n                title = df1.loc[i, 'Title']\n                df1.loc[i, 'Age'] = round(self.means[title], 2)\n        \n        df1[\"AgeGroup\"] = pd.cut(df1[\"Age\"], bins=[-0.001, 16.336, 32.252, 48.168, 64.084, 80.0], labels=[0,1,2,3,4])\n        df1[\"AgeGroup\"] = df1[\"AgeGroup\"].astype(int)\n\n        \n        return df1","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.407929Z","iopub.execute_input":"2022-07-22T11:08:54.408224Z","iopub.status.idle":"2022-07-22T11:08:54.421229Z","shell.execute_reply.started":"2022-07-22T11:08:54.408195Z","shell.execute_reply":"2022-07-22T11:08:54.419829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AgeTransformer().fit_transform(CustomAttributeTitle().fit_transform(data)).head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.423058Z","iopub.execute_input":"2022-07-22T11:08:54.423650Z","iopub.status.idle":"2022-07-22T11:08:54.543528Z","shell.execute_reply.started":"2022-07-22T11:08:54.423600Z","shell.execute_reply":"2022-07-22T11:08:54.542584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Custom Attributes","metadata":{}},{"cell_type":"markdown","source":"### Implementing Feature extraction to create new features\n* add SibSP and Parch attributes to get family belongings or followers on the ship","metadata":{}},{"cell_type":"code","source":"class CustomAttributes(BaseEstimator, TransformerMixin):\n    def fit(self, df):\n        return self\n    \n    def transform(self, df):\n        df1 = df.copy()\n        df1['FamilySize'] = df1['SibSp'] + df1['Parch']\n        df1 = df1.drop(['SibSp','Parch'], axis=1)\n        \n        df1['IsAlone'] = 0\n        df1.loc[df1['FamilySize'] == 1, 'IsAlone'] = 1\n        # df1 = df1.drop('FamilySize', axis=1)\n        \n        df1['Age*Class'] = df1.Age * df1.Pclass\n        \n        return df1","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.545234Z","iopub.execute_input":"2022-07-22T11:08:54.545568Z","iopub.status.idle":"2022-07-22T11:08:54.554230Z","shell.execute_reply.started":"2022-07-22T11:08:54.545534Z","shell.execute_reply":"2022-07-22T11:08:54.553090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CustomAttributes().fit_transform(data).head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.555947Z","iopub.execute_input":"2022-07-22T11:08:54.556264Z","iopub.status.idle":"2022-07-22T11:08:54.583986Z","shell.execute_reply.started":"2022-07-22T11:08:54.556233Z","shell.execute_reply":"2022-07-22T11:08:54.583146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='IsAlone',hue = 'Survived', data=CustomAttributes().fit_transform(data))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.585208Z","iopub.execute_input":"2022-07-22T11:08:54.585671Z","iopub.status.idle":"2022-07-22T11:08:54.735747Z","shell.execute_reply.started":"2022-07-22T11:08:54.585614Z","shell.execute_reply":"2022-07-22T11:08:54.734972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fare","metadata":{}},{"cell_type":"code","source":"plt.hist(data['Fare'])[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.736918Z","iopub.execute_input":"2022-07-22T11:08:54.737374Z","iopub.status.idle":"2022-07-22T11:08:54.911550Z","shell.execute_reply.started":"2022-07-22T11:08:54.737327Z","shell.execute_reply":"2022-07-22T11:08:54.910767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def FareBand(df):\n    df1 = df.copy()\n    df1['FareBand'] = pd.qcut(df1['Fare'], 4)\n    return df1[['FareBand', 'Survived']].groupby(['FareBand'], as_index=False).mean().sort_values(by='FareBand', ascending=True)\n\nFareBand(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.913010Z","iopub.execute_input":"2022-07-22T11:08:54.913633Z","iopub.status.idle":"2022-07-22T11:08:54.939235Z","shell.execute_reply.started":"2022-07-22T11:08:54.913583Z","shell.execute_reply":"2022-07-22T11:08:54.938123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Implementing Fare Column Transformer\n","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nclass FareTransformer(BaseEstimator, TransformerMixin):\n    def fit(self, df):\n        return self\n    \n    def transform(self, df):\n        df1 = df.copy()\n        # handle missing values\n        df1['Fare'] = SimpleImputer(strategy='mean').fit_transform(df1[['Fare']])\n        # Making bins according to distribution\n        df1[\"Fare\"] = pd.cut(df1[\"Fare\"], bins=[-0.001, 7.91, 14.454, 31.0, 512.329200], labels=[0,1,2,3])\n        df1[\"Fare\"] = df1[\"Fare\"].astype(int)\n        return df1","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.940743Z","iopub.execute_input":"2022-07-22T11:08:54.941381Z","iopub.status.idle":"2022-07-22T11:08:54.952167Z","shell.execute_reply.started":"2022-07-22T11:08:54.941328Z","shell.execute_reply":"2022-07-22T11:08:54.951125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FareTransformer().fit_transform(data).head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.956057Z","iopub.execute_input":"2022-07-22T11:08:54.956515Z","iopub.status.idle":"2022-07-22T11:08:54.991257Z","shell.execute_reply.started":"2022-07-22T11:08:54.956468Z","shell.execute_reply":"2022-07-22T11:08:54.990352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Fare', hue='Survived', data=FareTransformer().transform(data))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:54.992310Z","iopub.execute_input":"2022-07-22T11:08:54.992723Z","iopub.status.idle":"2022-07-22T11:08:55.172644Z","shell.execute_reply.started":"2022-07-22T11:08:54.992692Z","shell.execute_reply":"2022-07-22T11:08:55.171525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical Transformation","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nclass CategoricalTransformer(BaseEstimator, TransformerMixin):\n    def fit(self, df):\n        return self\n    \n    def transform(self, df):\n        df1 = df.copy()\n        df1[['Sex','Embarked','Pclass']] = SimpleImputer(strategy='most_frequent').fit_transform(df1[['Sex','Embarked','Pclass']])\n        df1['Sex'] = df1['Sex'].map({'male':0, 'female':1})\n        df1[['Embarked','Pclass']] = OrdinalEncoder().fit_transform(df1[['Embarked','Pclass']])\n        return df1","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.174704Z","iopub.execute_input":"2022-07-22T11:08:55.175092Z","iopub.status.idle":"2022-07-22T11:08:55.184381Z","shell.execute_reply.started":"2022-07-22T11:08:55.175059Z","shell.execute_reply":"2022-07-22T11:08:55.183226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CategoricalTransformer().fit_transform(data).head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.185699Z","iopub.execute_input":"2022-07-22T11:08:55.186026Z","iopub.status.idle":"2022-07-22T11:08:55.227828Z","shell.execute_reply.started":"2022-07-22T11:08:55.185996Z","shell.execute_reply":"2022-07-22T11:08:55.226801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transformation Pipeline","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.decomposition import PCA\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.229497Z","iopub.execute_input":"2022-07-22T11:08:55.230006Z","iopub.status.idle":"2022-07-22T11:08:55.236227Z","shell.execute_reply.started":"2022-07-22T11:08:55.229951Z","shell.execute_reply":"2022-07-22T11:08:55.234966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = data['Survived']\ndata = data.drop('Survived', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.237907Z","iopub.execute_input":"2022-07-22T11:08:55.238378Z","iopub.status.idle":"2022-07-22T11:08:55.249231Z","shell.execute_reply.started":"2022-07-22T11:08:55.238331Z","shell.execute_reply":"2022-07-22T11:08:55.247982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocess_pipeline = Pipeline([\n    ('title', CustomAttributeTitle()),\n    ('age', AgeTransformer()),\n    ('cabin', CabinTransformer()),\n    ('followers', CustomAttributes()),\n    ('fare', FareTransformer()),\n    ('encoding', CategoricalTransformer()),\n    ('pca', PCA(n_components=7))\n])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.250831Z","iopub.execute_input":"2022-07-22T11:08:55.251275Z","iopub.status.idle":"2022-07-22T11:08:55.261101Z","shell.execute_reply.started":"2022-07-22T11:08:55.251218Z","shell.execute_reply":"2022-07-22T11:08:55.259865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.262935Z","iopub.execute_input":"2022-07-22T11:08:55.263460Z","iopub.status.idle":"2022-07-22T11:08:55.286445Z","shell.execute_reply.started":"2022-07-22T11:08:55.263394Z","shell.execute_reply":"2022-07-22T11:08:55.285448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_processed = preprocess_pipeline.fit_transform(data)\nprint('shape of processed data:', data_processed.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.287772Z","iopub.execute_input":"2022-07-22T11:08:55.288112Z","iopub.status.idle":"2022-07-22T11:08:55.424929Z","shell.execute_reply.started":"2022-07-22T11:08:55.288081Z","shell.execute_reply":"2022-07-22T11:08:55.423862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Selection","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(data_processed, labels, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.426950Z","iopub.execute_input":"2022-07-22T11:08:55.427445Z","iopub.status.idle":"2022-07-22T11:08:55.437536Z","shell.execute_reply.started":"2022-07-22T11:08:55.427397Z","shell.execute_reply":"2022-07-22T11:08:55.436314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier, plot_tree\nfrom sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier, GradientBoostingClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import cross_val_score, GridSearchCV, RandomizedSearchCV\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.438870Z","iopub.execute_input":"2022-07-22T11:08:55.439240Z","iopub.status.idle":"2022-07-22T11:08:55.451340Z","shell.execute_reply.started":"2022-07-22T11:08:55.439205Z","shell.execute_reply":"2022-07-22T11:08:55.450195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {'LogReg': LogisticRegression(),\n         'XGBoost': XGBClassifier(),\n         'DecisionTree': DecisionTreeClassifier(),\n         'RandomForest': RandomForestClassifier(n_estimators=100),\n          'AdaBoost': AdaBoostClassifier(),\n          'GradBoost': GradientBoostingClassifier(),\n         'KNN': KNeighborsClassifier(n_neighbors=3),\n         'LinearSVC': LinearSVC(max_iter=1500)}\n\nres1, res2 = [], []\nfor i in models.keys():\n    model = models[i]\n    model.fit(x_train, y_train)\n    res1.append(model.score(x_train, y_train))\n    res2.append(cross_val_score(model, x_train, y_train).mean())\npd.DataFrame({'Models':models.keys(), 'Training Score':res1, 'Cross val Score':res2})","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:55.452853Z","iopub.execute_input":"2022-07-22T11:08:55.453199Z","iopub.status.idle":"2022-07-22T11:08:59.940589Z","shell.execute_reply.started":"2022-07-22T11:08:55.453165Z","shell.execute_reply":"2022-07-22T11:08:59.939350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic Regression","metadata":{}},{"cell_type":"code","source":"param_grid = [\n    {'penalty':['l2', 'none'],\n    'solver':['saga', 'lbfgs'],\n    'max_iter': [100,300, 500,1000]}\n]\n\nlr = LogisticRegression()\ngrid_search = GridSearchCV(lr, param_grid, cv=5)\ngrid_search.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:08:59.943832Z","iopub.execute_input":"2022-07-22T11:08:59.944376Z","iopub.status.idle":"2022-07-22T11:09:02.649005Z","shell.execute_reply.started":"2022-07-22T11:08:59.944322Z","shell.execute_reply":"2022-07-22T11:09:02.648118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:09:02.650421Z","iopub.execute_input":"2022-07-22T11:09:02.650756Z","iopub.status.idle":"2022-07-22T11:09:02.657012Z","shell.execute_reply.started":"2022-07-22T11:09:02.650722Z","shell.execute_reply":"2022-07-22T11:09:02.655950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = LogisticRegression(max_iter=500, penalty='none', solver='saga')\ncv = cross_val_score(lr, x_train, y_train, cv=5)\nprint(cv, '\\nmean:', cv.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:09:02.658291Z","iopub.execute_input":"2022-07-22T11:09:02.658649Z","iopub.status.idle":"2022-07-22T11:09:02.898697Z","shell.execute_reply.started":"2022-07-22T11:09:02.658588Z","shell.execute_reply":"2022-07-22T11:09:02.897775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Decision Tree","metadata":{}},{"cell_type":"code","source":"param_grid = [{\n    'criterion':['gini','entropy'], 'max_depth':[2,4,6,10], 'min_samples_leaf':[5,10,20,50,100], 'max_features':[None, 'log2','sqrt']\n}]\ntree = DecisionTreeClassifier()\ngrid_search = GridSearchCV(tree, param_grid, cv=5)\ngrid_search.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:09:02.900409Z","iopub.execute_input":"2022-07-22T11:09:02.900794Z","iopub.status.idle":"2022-07-22T11:09:04.646122Z","shell.execute_reply.started":"2022-07-22T11:09:02.900758Z","shell.execute_reply":"2022-07-22T11:09:04.644621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:09:04.649506Z","iopub.execute_input":"2022-07-22T11:09:04.649943Z","iopub.status.idle":"2022-07-22T11:09:04.656974Z","shell.execute_reply.started":"2022-07-22T11:09:04.649906Z","shell.execute_reply":"2022-07-22T11:09:04.655613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = DecisionTreeClassifier(max_depth=6, min_samples_leaf=10, criterion='entropy')\ncv = cross_val_score(lr, x_train, y_train, cv=5)\nprint(cv, '\\nmean:', cv.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:09:04.658699Z","iopub.execute_input":"2022-07-22T11:09:04.659168Z","iopub.status.idle":"2022-07-22T11:09:04.706723Z","shell.execute_reply.started":"2022-07-22T11:09:04.659122Z","shell.execute_reply":"2022-07-22T11:09:04.705518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RandomForest","metadata":{}},{"cell_type":"code","source":"param_grid = [{\n    'n_estimators': [100, 200, 300],\n    'max_features': [None, \"sqrt\", \"log2\"],\n    'max_depth': [3,5,7,10,12],\n    'min_samples_split': [10, 20],\n    'min_samples_leaf': [5],\n}]\n\nforest = RandomForestClassifier()\ngrid_search = GridSearchCV(forest, param_grid, cv=5, verbose=1, n_jobs=-1)\ngrid_search.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:09:04.708574Z","iopub.execute_input":"2022-07-22T11:09:04.709064Z","iopub.status.idle":"2022-07-22T11:10:47.454361Z","shell.execute_reply.started":"2022-07-22T11:09:04.709014Z","shell.execute_reply":"2022-07-22T11:10:47.453022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:10:47.456671Z","iopub.execute_input":"2022-07-22T11:10:47.457181Z","iopub.status.idle":"2022-07-22T11:10:47.464937Z","shell.execute_reply.started":"2022-07-22T11:10:47.457129Z","shell.execute_reply":"2022-07-22T11:10:47.464094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"forest = RandomForestClassifier(n_estimators=200, min_samples_split=10, min_samples_leaf=5, max_features=None, max_depth=7, bootstrap=True)\ncv = cross_val_score(forest, x_train, y_train, cv=5)\nprint(cv, '\\nmean:', cv.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:10:47.466341Z","iopub.execute_input":"2022-07-22T11:10:47.467344Z","iopub.status.idle":"2022-07-22T11:10:50.862724Z","shell.execute_reply.started":"2022-07-22T11:10:47.467301Z","shell.execute_reply":"2022-07-22T11:10:50.861837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Competition Submission","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/titanic/test.csv')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:13:41.675576Z","iopub.execute_input":"2022-07-22T11:13:41.675981Z","iopub.status.idle":"2022-07-22T11:13:41.700771Z","shell.execute_reply.started":"2022-07-22T11:13:41.675946Z","shell.execute_reply":"2022-07-22T11:13:41.699646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RandomForestClassifier(**grid_search.best_params_)\nmodel.fit(data_processed, labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:13:54.460716Z","iopub.execute_input":"2022-07-22T11:13:54.461462Z","iopub.status.idle":"2022-07-22T11:13:54.933288Z","shell.execute_reply.started":"2022-07-22T11:13:54.461405Z","shell.execute_reply":"2022-07-22T11:13:54.932093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test.drop(['PassengerId','Ticket'], axis=1)\ntest_processed = preprocess_pipeline.transform(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:13:56.929561Z","iopub.execute_input":"2022-07-22T11:13:56.929989Z","iopub.status.idle":"2022-07-22T11:13:57.024853Z","shell.execute_reply.started":"2022-07-22T11:13:56.929955Z","shell.execute_reply":"2022-07-22T11:13:57.023774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_processed)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:14:01.736018Z","iopub.execute_input":"2022-07-22T11:14:01.736507Z","iopub.status.idle":"2022-07-22T11:14:01.763217Z","shell.execute_reply.started":"2022-07-22T11:14:01.736466Z","shell.execute_reply":"2022-07-22T11:14:01.761868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'PassengerId':test['PassengerId'], 'Survived':pred})\nsubmission.to_csv('Submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:14:07.305024Z","iopub.execute_input":"2022-07-22T11:14:07.305401Z","iopub.status.idle":"2022-07-22T11:14:07.313046Z","shell.execute_reply.started":"2022-07-22T11:14:07.305368Z","shell.execute_reply":"2022-07-22T11:14:07.311772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('Submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:14:10.799943Z","iopub.execute_input":"2022-07-22T11:14:10.800330Z","iopub.status.idle":"2022-07-22T11:14:10.817679Z","shell.execute_reply.started":"2022-07-22T11:14:10.800298Z","shell.execute_reply":"2022-07-22T11:14:10.816231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}