{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport pickle\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-04T15:30:15.689858Z","iopub.execute_input":"2022-07-04T15:30:15.690326Z","iopub.status.idle":"2022-07-04T15:30:16.405205Z","shell.execute_reply.started":"2022-07-04T15:30:15.690290Z","shell.execute_reply":"2022-07-04T15:30:16.404166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing various sklearn packages","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import cross_validate\n\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\n\n# Preprocessing\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\n\n# dimensionality reduction\nfrom sklearn.decomposition import PCA\n\n# Hyperparameters Search \nfrom sklearn.model_selection import RandomizedSearchCV\n\n# To handle imbalanced ds\nfrom sklearn.utils import class_weight\n\n# Classification\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nfrom xgboost import XGBClassifier\n\n## Feature Selection RFE\nfrom sklearn.feature_selection import RFE\nfrom sklearn.feature_selection import RFECV\n\n## Isolation Forest - outliers detection\nfrom sklearn.ensemble import IsolationForest\n\n## Metrics\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:17.148228Z","iopub.execute_input":"2022-07-04T15:30:17.149085Z","iopub.status.idle":"2022-07-04T15:30:17.669469Z","shell.execute_reply.started":"2022-07-04T15:30:17.149050Z","shell.execute_reply":"2022-07-04T15:30:17.667841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading Data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/titanic/train.csv\") # reading the csvs\ndf_test = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ndf_sub_gen = pd.read_csv(\"/kaggle/input/titanic/gender_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:18.147419Z","iopub.execute_input":"2022-07-04T15:30:18.147702Z","iopub.status.idle":"2022-07-04T15:30:18.185175Z","shell.execute_reply.started":"2022-07-04T15:30:18.147680Z","shell.execute_reply":"2022-07-04T15:30:18.184216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:18.571503Z","iopub.execute_input":"2022-07-04T15:30:18.571825Z","iopub.status.idle":"2022-07-04T15:30:18.598916Z","shell.execute_reply.started":"2022-07-04T15:30:18.571801Z","shell.execute_reply":"2022-07-04T15:30:18.598054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:19.477185Z","iopub.execute_input":"2022-07-04T15:30:19.477717Z","iopub.status.idle":"2022-07-04T15:30:19.517785Z","shell.execute_reply.started":"2022-07-04T15:30:19.477689Z","shell.execute_reply":"2022-07-04T15:30:19.517035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:19.742356Z","iopub.execute_input":"2022-07-04T15:30:19.742896Z","iopub.status.idle":"2022-07-04T15:30:19.759784Z","shell.execute_reply.started":"2022-07-04T15:30:19.742859Z","shell.execute_reply":"2022-07-04T15:30:19.759029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:20.036226Z","iopub.execute_input":"2022-07-04T15:30:20.037051Z","iopub.status.idle":"2022-07-04T15:30:20.050763Z","shell.execute_reply.started":"2022-07-04T15:30:20.037020Z","shell.execute_reply":"2022-07-04T15:30:20.049670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Note that columns Age, Cabin and Embarked have nans, will need to fill them. ","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:20.245099Z","iopub.execute_input":"2022-07-04T15:30:20.245485Z","iopub.status.idle":"2022-07-04T15:30:20.252743Z","shell.execute_reply.started":"2022-07-04T15:30:20.245456Z","shell.execute_reply":"2022-07-04T15:30:20.251064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data preprocessing","metadata":{}},{"cell_type":"code","source":"class DataPreparation:\n    \n    def __init__(\n        self,\n        df_train: pd.DataFrame,\n        df_test: pd.DataFrame\n    ):\n        self.df_train = df_train\n        self.df_test = df_test\n        self.add_sex()\n        self.add_title()\n        self.add_embarked()\n        self.add_cabin()\n        \n    def add_sex(self):\n        all_sex = pd.concat([self.df_train.Sex, self.df_test.Sex])\n        self.transform_labels(all_sex, \"Sex\")\n        \n    def add_title(self, min_occurences=8):\n        train_title = self.get_title(self.df_train.Name)\n        test_title = self.get_title(self.df_test.Name)\n        self.df_train[\"title\"] = train_title\n        self.df_test[\"title\"] = test_title\n        all_titles = pd.concat([train_title, test_title])\n        dict_of_titles = all_titles.value_counts() # count occurencies of different titles\n        rare_titles = [key for key, val in dict_of_titles.items() if val <= min_occurences] # rare titles\n        dict_keys = rare_titles\n        dict_vals = len(rare_titles)*[\"rare\"]\n        mapping_dict_rare_titles = dict(zip(dict_keys, dict_vals))\n        self.df_train.replace({\"title\": mapping_dict_rare_titles}, inplace=True)\n        self.df_test.replace({\"title\": mapping_dict_rare_titles}, inplace=True)\n        all_titles = pd.concat([self.df_train.title, self.df_test.title])\n        self.transform_labels(all_titles, \"title\")\n        \n    def add_embarked(self):\n        all_embarked = pd.concat([self.df_train.Embarked, self.df_test.Embarked])\n        most_commont_value_emb= all_embarked.mode()[0] \n        self.df_train.Embarked.fillna(most_commont_value_emb,  inplace=True) # filling the nans with the most common value in Embarked\n        self.df_test.Embarked.fillna(most_commont_value_emb,  inplace=True)\n        self.onehotencoder_labels(\"Embarked\")\n          \n    def add_cabin(self):\n        cabin_char_train = self.df_train.Cabin.str[0]\n        cabin_char_test = self.df_test.Cabin.str[0]\n        self.df_train[\"CabIN\"] = self.df_train.Cabin.str[0]\n        self.df_test[\"CabIN\"] = self.df_test.Cabin.str[0]\n        all_cabin_char = pd.concat([cabin_char_train, cabin_char_test])\n        most_commont_value_cabin= all_cabin_char.mode()[0]\n        self.transform_labels(all_cabin_char, \"CabIN\")\n        self.df_train.CabIN.fillna(most_commont_value_cabin, inplace=True) # filling the nans with the most common value in Cabin\n        self.df_test.CabIN.fillna(most_commont_value_cabin, inplace=True)\n        \n    def create_num_df(self):\n        df_train_num = self.df_train.select_dtypes(include=[\"number\"]) # Selecting only numerical columns\n        df_test_num = self.df_test.select_dtypes(include=[\"number\"])\n        return df_train_num, df_test_num\n        \n    def transform_labels(self, df_column: pd.Series, name: str):\n        label_encoder = LabelEncoder().fit(df_column)\n        self.df_train[f\"{name}_num\"] = label_encoder.transform(self.df_train[name])\n        self.df_test[f\"{name}_num\"] = label_encoder.transform(self.df_test[name])\n        \n    def onehotencoder_labels(self, column: str):\n        one_hot_encoder_train = pd.get_dummies(self.df_train[column])\n        one_hot_encoder_test = pd.get_dummies(self.df_test[column])\n        self.df_train = pd.concat([self.df_train, one_hot_encoder_train], axis=1)\n        self.df_test = pd.concat([self.df_test, one_hot_encoder_test], axis=1)\n     \n    @staticmethod\n    def get_title(col_name: str) -> pd.Series:\n        first_split = col_name.str.split(\".\").str[0]\n        second_split = first_split.str.split(\",\").str[1]\n        return second_split","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:20.727967Z","iopub.execute_input":"2022-07-04T15:30:20.729120Z","iopub.status.idle":"2022-07-04T15:30:20.756794Z","shell.execute_reply.started":"2022-07-04T15:30:20.729069Z","shell.execute_reply":"2022-07-04T15:30:20.755840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_preparation = DataPreparation(df_train, df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:20.989510Z","iopub.execute_input":"2022-07-04T15:30:20.989832Z","iopub.status.idle":"2022-07-04T15:30:21.027623Z","shell.execute_reply.started":"2022-07-04T15:30:20.989808Z","shell.execute_reply":"2022-07-04T15:30:21.026293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_num, df_test_num = data_preparation.create_num_df()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:21.238186Z","iopub.execute_input":"2022-07-04T15:30:21.238567Z","iopub.status.idle":"2022-07-04T15:30:21.251004Z","shell.execute_reply.started":"2022-07-04T15:30:21.238538Z","shell.execute_reply":"2022-07-04T15:30:21.249788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(df_train_num.Survived != df_train_num.Sex_num).sum() / df_train_num.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:21.479201Z","iopub.execute_input":"2022-07-04T15:30:21.479724Z","iopub.status.idle":"2022-07-04T15:30:21.487813Z","shell.execute_reply.started":"2022-07-04T15:30:21.479695Z","shell.execute_reply":"2022-07-04T15:30:21.486926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_num.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:21.701694Z","iopub.execute_input":"2022-07-04T15:30:21.702339Z","iopub.status.idle":"2022-07-04T15:30:21.720367Z","shell.execute_reply.started":"2022-07-04T15:30:21.702300Z","shell.execute_reply":"2022-07-04T15:30:21.719041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualisation","metadata":{}},{"cell_type":"code","source":"corr_matrix = df_train_num.corr()\nfig = plt.figure(figsize=(10, 10))\ndisplay(corr_matrix)\nsns.heatmap(corr_matrix);","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:22.198108Z","iopub.execute_input":"2022-07-04T15:30:22.198447Z","iopub.status.idle":"2022-07-04T15:30:22.540131Z","shell.execute_reply.started":"2022-07-04T15:30:22.198420Z","shell.execute_reply":"2022-07-04T15:30:22.538828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Features sorted by correlation\nnp.abs(corr_matrix[\"Survived\"]).sort_values()\n# we remove PassengerID as it has very low correlation with Survived","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:22.541704Z","iopub.execute_input":"2022-07-04T15:30:22.542182Z","iopub.status.idle":"2022-07-04T15:30:22.551008Z","shell.execute_reply.started":"2022-07-04T15:30:22.542158Z","shell.execute_reply":"2022-07-04T15:30:22.549969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_num.drop(columns=[\"PassengerId\"], inplace=True)\ndf_test_num.drop(columns=[\"PassengerId\"], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:22.773142Z","iopub.execute_input":"2022-07-04T15:30:22.773531Z","iopub.status.idle":"2022-07-04T15:30:22.778155Z","shell.execute_reply.started":"2022-07-04T15:30:22.773503Z","shell.execute_reply":"2022-07-04T15:30:22.777043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_num.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:23.207235Z","iopub.execute_input":"2022-07-04T15:30:23.207794Z","iopub.status.idle":"2022-07-04T15:30:23.223445Z","shell.execute_reply.started":"2022-07-04T15:30:23.207764Z","shell.execute_reply":"2022-07-04T15:30:23.222486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.plotting.scatter_matrix(df_train, figsize=(10,10), diagonal=\"kde\") # kernel estimation\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:24.014166Z","iopub.execute_input":"2022-07-04T15:30:24.015367Z","iopub.status.idle":"2022-07-04T15:30:27.689909Z","shell.execute_reply.started":"2022-07-04T15:30:24.015317Z","shell.execute_reply":"2022-07-04T15:30:27.688675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,10))\nsns.barplot(data=df_train[[\"Embarked\", \"Survived\"]] , x='Embarked', y='Survived'); # embarked versus survival","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:27.692239Z","iopub.execute_input":"2022-07-04T15:30:27.693334Z","iopub.status.idle":"2022-07-04T15:30:27.907943Z","shell.execute_reply.started":"2022-07-04T15:30:27.693294Z","shell.execute_reply":"2022-07-04T15:30:27.907034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,10))\nsns.barplot(data=df_train[[\"Sex\", \"Survived\"]] , x='Sex', y='Survived'); # Sex versus survival","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:27.909300Z","iopub.execute_input":"2022-07-04T15:30:27.909815Z","iopub.status.idle":"2022-07-04T15:30:28.153622Z","shell.execute_reply.started":"2022-07-04T15:30:27.909780Z","shell.execute_reply":"2022-07-04T15:30:28.152825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,10))\nsns.barplot(data=df_train[[\"title\", \"Survived\"]] , x='title', y='Survived'); # title versus survival","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:28.155636Z","iopub.execute_input":"2022-07-04T15:30:28.156316Z","iopub.status.idle":"2022-07-04T15:30:28.482661Z","shell.execute_reply.started":"2022-07-04T15:30:28.156286Z","shell.execute_reply":"2022-07-04T15:30:28.481660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Detecting Outliers","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2,figsize=(15,5))\nbox1 = df_train.Fare.plot.box(ax=ax[0]);\nbox2 = df_train.Age.plot.box(ax=ax[1]);","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:28.483890Z","iopub.execute_input":"2022-07-04T15:30:28.484214Z","iopub.status.idle":"2022-07-04T15:30:28.734935Z","shell.execute_reply.started":"2022-07-04T15:30:28.484187Z","shell.execute_reply":"2022-07-04T15:30:28.734159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Get outliers\n\ndef get_outliers(df, column: pd.Series):\n    first_quantile = df[column].quantile(0.25)\n    third_quantile = df[column].quantile(0.75)\n    IQR = third_quantile - first_quantile \n    outlier_threshold = third_quantile + 1.5*IQR\n    return outlier_threshold\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:28.736016Z","iopub.execute_input":"2022-07-04T15:30:28.736481Z","iopub.status.idle":"2022-07-04T15:30:28.742249Z","shell.execute_reply.started":"2022-07-04T15:30:28.736450Z","shell.execute_reply":"2022-07-04T15:30:28.741431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_outlier = get_outliers(df_train, \"Age\")\nfare_outlier = get_outliers(df_train, \"Fare\")\nage_outlier, fare_outlier","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:28.743283Z","iopub.execute_input":"2022-07-04T15:30:28.744258Z","iopub.status.idle":"2022-07-04T15:30:28.768313Z","shell.execute_reply.started":"2022-07-04T15:30:28.744227Z","shell.execute_reply":"2022-07-04T15:30:28.767254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_num = df_train_num[~((df_train_num.Age >= age_outlier) & (df_train_num.Fare >= fare_outlier))]","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:28.769576Z","iopub.execute_input":"2022-07-04T15:30:28.769845Z","iopub.status.idle":"2022-07-04T15:30:28.780017Z","shell.execute_reply.started":"2022-07-04T15:30:28.769818Z","shell.execute_reply":"2022-07-04T15:30:28.779028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ML Models","metadata":{}},{"cell_type":"code","source":"X = df_train_num.loc[:, df_train_num.columns != 'Survived']\ny = df_train_num.Survived","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:29.461210Z","iopub.execute_input":"2022-07-04T15:30:29.461591Z","iopub.status.idle":"2022-07-04T15:30:29.467967Z","shell.execute_reply.started":"2022-07-04T15:30:29.461562Z","shell.execute_reply":"2022-07-04T15:30:29.466961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation Split\nseed = 37\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.15, random_state=seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:29.813799Z","iopub.execute_input":"2022-07-04T15:30:29.814130Z","iopub.status.idle":"2022-07-04T15:30:29.822745Z","shell.execute_reply.started":"2022-07-04T15:30:29.814106Z","shell.execute_reply":"2022-07-04T15:30:29.821681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# processing num data to eliminate nans\nnumerical_transformer = SimpleImputer(strategy='median')\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, X.columns),\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:31.270381Z","iopub.execute_input":"2022-07-04T15:30:31.270743Z","iopub.status.idle":"2022-07-04T15:30:31.276535Z","shell.execute_reply.started":"2022-07-04T15:30:31.270715Z","shell.execute_reply":"2022-07-04T15:30:31.275128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# classes_weights = class_weight.compute_sample_weight(\n#     class_weight='balanced',\n#     y=y_train\n# )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Baseline model aka Gender Submission","metadata":{}},{"cell_type":"code","source":"(X_train.Sex_num != y_train).sum() / y_train.shape[0], (X_val.Sex_num != y_val).sum() / y_val.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:36:05.722162Z","iopub.execute_input":"2022-07-04T15:36:05.722510Z","iopub.status.idle":"2022-07-04T15:36:05.731363Z","shell.execute_reply.started":"2022-07-04T15:36:05.722483Z","shell.execute_reply":"2022-07-04T15:36:05.730383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic Regression","metadata":{}},{"cell_type":"code","source":"clf1 = LogisticRegression() # making a baseline model Random Forest (RF)\nscaler = MinMaxScaler()\npipeline1 = Pipeline([('preprocess', preprocessor), ('scaler', scaler), ('clf', clf1)]) # Making a pipeline\npipeline1.fit(X_train, y_train)#, clf__sample_weight=classes_weights) # Fitting the model\ny_pred = pipeline1.predict(X_val) # Obtaining predictions on validation part that the model has never seen\naccuracy_score(y_val, y_pred), f1_score(y_val, y_pred) # Accuracy, F1 Score (Precision and Recall indicator), Out of Bag Score (only in RF)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:39.698360Z","iopub.execute_input":"2022-07-04T15:30:39.698800Z","iopub.status.idle":"2022-07-04T15:30:39.734178Z","shell.execute_reply.started":"2022-07-04T15:30:39.698769Z","shell.execute_reply":"2022-07-04T15:30:39.733353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest CLF","metadata":{}},{"cell_type":"code","source":"clf2 = RandomForestClassifier(oob_score=True, random_state=seed) # making a baseline model Random Forest (RF)\nscaler = MinMaxScaler()\npipeline2 = Pipeline([('preprocess', preprocessor), ('scaler', scaler), ('clf', clf2)]) # Making a pipeline\npipeline2.fit(X_train, y_train)#, clf__sample_weight=classes_weights) # Fitting the model\ny_pred = pipeline2.predict(X_val) # Obtaining predictions on validation part that the model has never seen\naccuracy_score(y_val, y_pred), f1_score(y_val, y_pred), pipeline2[\"clf\"].oob_score_ ","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:40.738646Z","iopub.execute_input":"2022-07-04T15:30:40.739800Z","iopub.status.idle":"2022-07-04T15:30:40.985819Z","shell.execute_reply.started":"2022-07-04T15:30:40.739758Z","shell.execute_reply":"2022-07-04T15:30:40.984818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGBoost CLF","metadata":{}},{"cell_type":"code","source":"clf3 = XGBClassifier()\nscaler = MinMaxScaler()\npipeline3 = Pipeline([('preprocess', preprocessor), ('scaler', scaler), ('clf', clf3)]) \npipeline3.fit(X_train, y_train)#, clf__sample_weight=classes_weights)\ny_pred = pipeline3.predict(X_val) \naccuracy_score(y_val, y_pred), f1_score(y_val, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:42.113070Z","iopub.execute_input":"2022-07-04T15:30:42.113392Z","iopub.status.idle":"2022-07-04T15:30:42.456434Z","shell.execute_reply.started":"2022-07-04T15:30:42.113368Z","shell.execute_reply":"2022-07-04T15:30:42.455627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hyperparameter tuning of RF model","metadata":{}},{"cell_type":"code","source":"# Number of trees in the random \nn_estimators = [128, 256]\n# Number of features to consider at every split\nmax_features = ['auto', 'sqrt'] + list(np.arange(0.5, 1.0, 0.1))\n# Maximum number of levels in tree\nmax_depth = [None, 6, 8]\n# Minimum number of samples required to split a node\nmin_samples_split = [4, 6, 8]\n# Minimum number of samples required at each leaf node\nmin_samples_leaf = [2, 4, 6]\n# Method of selecting samples for training each tree\nbootstrap = [True]","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:43.418498Z","iopub.execute_input":"2022-07-04T15:30:43.418853Z","iopub.status.idle":"2022-07-04T15:30:43.425362Z","shell.execute_reply.started":"2022-07-04T15:30:43.418824Z","shell.execute_reply":"2022-07-04T15:30:43.424560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let us save the above choices in a dictionary\nparameter_selection = {\n    'clf__criterion': ['gini'],                 \n    'clf__n_estimators': n_estimators,\n    'clf__max_features': max_features,\n    'clf__max_depth': max_depth,\n    'clf__min_samples_split': min_samples_split,\n    'clf__min_samples_leaf': min_samples_leaf,\n    'clf__bootstrap': bootstrap\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:44.048403Z","iopub.execute_input":"2022-07-04T15:30:44.048720Z","iopub.status.idle":"2022-07-04T15:30:44.055184Z","shell.execute_reply.started":"2022-07-04T15:30:44.048696Z","shell.execute_reply":"2022-07-04T15:30:44.053798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = RandomForestClassifier(oob_score=True, random_state=seed)\nscaler = MinMaxScaler()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:44.562593Z","iopub.execute_input":"2022-07-04T15:30:44.563431Z","iopub.status.idle":"2022-07-04T15:30:44.567907Z","shell.execute_reply.started":"2022-07-04T15:30:44.563401Z","shell.execute_reply":"2022-07-04T15:30:44.566681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipeline = Pipeline([('preprocess', preprocessor), ('scaler', scaler), ('clf', clf)])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:44.927648Z","iopub.execute_input":"2022-07-04T15:30:44.928850Z","iopub.status.idle":"2022-07-04T15:30:44.934809Z","shell.execute_reply.started":"2022-07-04T15:30:44.928807Z","shell.execute_reply":"2022-07-04T15:30:44.933690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using Ranfom Search with 5-Fold Cross Validation\nk_fold = 5\nclf_rand_search = RandomizedSearchCV(estimator = pipeline, param_distributions = parameter_selection, n_iter = 100, cv = k_fold, verbose=1, random_state=seed, n_jobs = -1, scoring='accuracy')","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:30:57.707558Z","iopub.execute_input":"2022-07-04T15:30:57.707932Z","iopub.status.idle":"2022-07-04T15:30:57.714099Z","shell.execute_reply.started":"2022-07-04T15:30:57.707903Z","shell.execute_reply":"2022-07-04T15:30:57.712609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_rand_search.fit(X_train, y_train)#, clf__sample_weight=classes_weights) # performing random search of hyperparameters with cross validation, it takes a while","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:31:09.022075Z","iopub.execute_input":"2022-07-04T15:31:09.022460Z","iopub.status.idle":"2022-07-04T15:32:15.630184Z","shell.execute_reply.started":"2022-07-04T15:31:09.022432Z","shell.execute_reply":"2022-07-04T15:32:15.629213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtaining best parameters\nbest_params = clf_rand_search.best_params_\nprint(best_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:15.632297Z","iopub.execute_input":"2022-07-04T15:32:15.632658Z","iopub.status.idle":"2022-07-04T15:32:15.636425Z","shell.execute_reply.started":"2022-07-04T15:32:15.632636Z","shell.execute_reply":"2022-07-04T15:32:15.635808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validating the model","metadata":{}},{"cell_type":"code","source":"clf = RandomForestClassifier(oob_score=True, random_state=seed)\nscaler = MinMaxScaler()\npipeline = Pipeline([('preprocess', preprocessor), ('scaler', scaler), ('clf', clf)])\npipeline.set_params(**best_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:15.637202Z","iopub.execute_input":"2022-07-04T15:32:15.637737Z","iopub.status.idle":"2022-07-04T15:32:15.669481Z","shell.execute_reply.started":"2022-07-04T15:32:15.637714Z","shell.execute_reply":"2022-07-04T15:32:15.668028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipeline.fit(X_train, y_train)#, clf__sample_weight=classes_weights)\ny_pred = pipeline.predict(X_val)\naccuracy_score(y_val, y_pred), f1_score(y_val, y_pred), pipeline[\"clf\"].oob_score_","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:15.672249Z","iopub.execute_input":"2022-07-04T15:32:15.672583Z","iopub.status.idle":"2022-07-04T15:32:15.949529Z","shell.execute_reply.started":"2022-07-04T15:32:15.672556Z","shell.execute_reply":"2022-07-04T15:32:15.948465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing the submission","metadata":{}},{"cell_type":"code","source":"clf = RandomForestClassifier(oob_score=True, random_state=seed)\nscaler = MinMaxScaler()\npipeline = Pipeline([('preprocess', preprocessor), ('scaler', scaler), ('clf', clf)])\npipeline.set_params(**best_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:15.950725Z","iopub.execute_input":"2022-07-04T15:32:15.951041Z","iopub.status.idle":"2022-07-04T15:32:15.970101Z","shell.execute_reply.started":"2022-07-04T15:32:15.951010Z","shell.execute_reply":"2022-07-04T15:32:15.969014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving the model via pickle\nfilename = '/kaggle/working/finalized_model.sav'\npickle.dump(pipeline, open(filename, 'wb'))","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:15.971283Z","iopub.execute_input":"2022-07-04T15:32:15.972140Z","iopub.status.idle":"2022-07-04T15:32:15.981139Z","shell.execute_reply.started":"2022-07-04T15:32:15.972116Z","shell.execute_reply":"2022-07-04T15:32:15.980330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the model \nloaded_model = pickle.load(open(filename, 'rb'))","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:15.982232Z","iopub.execute_input":"2022-07-04T15:32:15.983116Z","iopub.status.idle":"2022-07-04T15:32:15.994605Z","shell.execute_reply.started":"2022-07-04T15:32:15.983053Z","shell.execute_reply":"2022-07-04T15:32:15.993852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loaded_model.fit(X_train, y_train)#, clf__sample_weight=classes_weights) \nloaded_model[\"clf\"].oob_score_ # Out of Bag Score is a Validation type Score","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:15.995769Z","iopub.execute_input":"2022-07-04T15:32:15.996287Z","iopub.status.idle":"2022-07-04T15:32:16.294155Z","shell.execute_reply.started":"2022-07-04T15:32:15.996258Z","shell.execute_reply":"2022-07-04T15:32:16.293216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub_gen","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:16.295319Z","iopub.execute_input":"2022-07-04T15:32:16.295740Z","iopub.status.idle":"2022-07-04T15:32:16.307293Z","shell.execute_reply.started":"2022-07-04T15:32:16.295713Z","shell.execute_reply":"2022-07-04T15:32:16.306138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame()\ndf_submission[\"PassengerId\"] = df_sub_gen.PassengerId","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:16.311290Z","iopub.execute_input":"2022-07-04T15:32:16.312226Z","iopub.status.idle":"2022-07-04T15:32:16.319177Z","shell.execute_reply.started":"2022-07-04T15:32:16.312187Z","shell.execute_reply":"2022-07-04T15:32:16.318417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission[\"Survived\"] = loaded_model.predict(df_test_num)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:16.320625Z","iopub.execute_input":"2022-07-04T15:32:16.320921Z","iopub.status.idle":"2022-07-04T15:32:16.356767Z","shell.execute_reply.started":"2022-07-04T15:32:16.320890Z","shell.execute_reply":"2022-07-04T15:32:16.355339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:16.358646Z","iopub.execute_input":"2022-07-04T15:32:16.358970Z","iopub.status.idle":"2022-07-04T15:32:16.367669Z","shell.execute_reply.started":"2022-07-04T15:32:16.358947Z","shell.execute_reply":"2022-07-04T15:32:16.366632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:32:16.369140Z","iopub.execute_input":"2022-07-04T15:32:16.369448Z","iopub.status.idle":"2022-07-04T15:32:16.387100Z","shell.execute_reply.started":"2022-07-04T15:32:16.369420Z","shell.execute_reply":"2022-07-04T15:32:16.385727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}