{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## DISCLAMER\nAlbeit stated in the title, getting an 100% accuracy should not be the end-goal of any real-world data science project. If your model scores a 100% accuracy, most probably you are overfitting or you have a data leakage on your hands. As the modern proverb goes, “If it seems too good to be true, it probably is.”\n\nThe main purpose of this notebook is to train myself various data wrangling techniques in order to get the best possible result fairly following best data science practices. At the same time, I hope this notebook would be deemed useful by the community especially for newcomers to start their data science journey. \n\nConstructive criticism are most welcome!\n\n### Solution inspired by these amazing people:\n- [MANAV SEHGAL](https://www.kaggle.com/code/startupsci/titanic-data-science-solutions/notebook)\n- [GUNES EVITAN](https://www.kaggle.com/code/gunesevitan/titanic-advanced-feature-engineering-tutorial/notebook)","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-24T01:52:29.954575Z","iopub.execute_input":"2022-07-24T01:52:29.955715Z","iopub.status.idle":"2022-07-24T01:52:29.983435Z","shell.execute_reply.started":"2022-07-24T01:52:29.955582Z","shell.execute_reply":"2022-07-24T01:52:29.982145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data analysis and wrangling\nimport numpy as np\nimport pandas as pd\nfrom scipy.stats import chi2_contingency\nfrom sklearn.impute import SimpleImputer\n\n# visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:52:29.986523Z","iopub.execute_input":"2022-07-24T01:52:29.986904Z","iopub.status.idle":"2022-07-24T01:52:31.230902Z","shell.execute_reply.started":"2022-07-24T01:52:29.986873Z","shell.execute_reply":"2022-07-24T01:52:31.229806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read files and have a glimpse of the data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/titanic/train.csv')\ntest = pd.read_csv('/kaggle/input/titanic/test.csv')\ncombine = [train, test]\nprint('Number of rows (Train): ' + str(len(train)))\nprint('Number of DUPLICATE rows (Train): ' + str(len(train) - len(train.drop_duplicates())))\nprint('_' * 40)\nprint('Number of rows (Test): ' + str(len(test)))\nprint('Number of DUPLICATE rows (Test): ' + str(len(test) - len(test.drop_duplicates())))\ntrain[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:18.887857Z","iopub.execute_input":"2022-07-24T01:55:18.889163Z","iopub.status.idle":"2022-07-24T01:55:18.934993Z","shell.execute_reply.started":"2022-07-24T01:55:18.889119Z","shell.execute_reply":"2022-07-24T01:55:18.933692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# Check number of unique values for every column\nfor col in train.columns:\n    print(col + ': ' + str(train[col].nunique()) + ' unique values')\nprint('_'*80)\nprint(train.info())\nprint('_'*80)\nprint(train.describe())\nprint('_'*80)\nprint(train.describe(include=['O']))\nprint('_'*80)\n# Check for missing values in every column\nprint('Number of missing value(s) in every column (Train):')\nprint(train.isnull().sum())\nprint('Number of missing value(s) in every column (Test):')\nprint(test.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:20.994842Z","iopub.execute_input":"2022-07-24T01:55:20.996070Z","iopub.status.idle":"2022-07-24T01:55:21.063353Z","shell.execute_reply.started":"2022-07-24T01:55:20.996028Z","shell.execute_reply":"2022-07-24T01:55:21.061847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot charts between multiple variables\n# age\nfig, ax=plt.subplots(1,figsize=(8,6))\nsns.boxplot(x='Survived',y='Age', data=train)\nax.set_ylim(0,100)\nplt.title(\"Survived vs Age\")\nplt.show()\n\n# Sex\nfig, ax=plt.subplots(1,figsize=(8,6))\nsns.countplot(x='Survived' ,hue='Sex', data=train)\nax.set_ylim(0,500)\nplt.title(\"Survived vs Sex\")\nplt.show()\n\n# Pclass\nfig, ax=plt.subplots(1,figsize=(8,6))\nsns.countplot(x='Survived' ,hue='Pclass', data=train)\nax.set_ylim(0,400)\nplt.title(\"Survived vs Pclass\")\nplt.show()\n\n# Embarked\nfig, ax=plt.subplots(1,figsize=(8,6))\nsns.countplot(x='Survived' ,hue='Embarked', data=train)\nax.set_ylim(0,500)\nplt.title(\"Survived vs Embarked\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-24T01:55:21.226756Z","iopub.execute_input":"2022-07-24T01:55:21.227568Z","iopub.status.idle":"2022-07-24T01:55:22.123224Z","shell.execute_reply.started":"2022-07-24T01:55:21.227517Z","shell.execute_reply":"2022-07-24T01:55:22.121547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Insights from the charts:\n- For Age, few points are located outside the whiskers of the box plot. However, since it is still within a reasonable age range (around 62 - 80 years old) we'll keep the data\n- Females were more likely to survived than Male\n- Upper-class passengers (Pclass = 1) were more likely to survived that crash\n- Passengers embarked from S were less likely to survived (Possibility of a correlation between Pclass and Embarked)\n\n## Additional insights:\n- Since class of passengers is significant in predicting survivability, it might be worth trying to infer the social status of a passenger from his/her name\n- Since a lof of the values are unique, dropping the 'PassengerId', 'Ticket' and the 'Cabin' columns might improve the accuracy of our model\n- Might be worthwhile to combine 'SibSp' and 'Parch' columns to create 'FamilySize' column instead","metadata":{}},{"cell_type":"code","source":"# Drop PassengerId, Ticket and Cabin columns\ntrain = train.drop(['PassengerId', 'Ticket', 'Cabin'], axis = 1)\ntest = test.drop(['Ticket', 'Cabin'], axis = 1)\ncombine = [train, test]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:22.567335Z","iopub.execute_input":"2022-07-24T01:55:22.567770Z","iopub.status.idle":"2022-07-24T01:55:22.577503Z","shell.execute_reply.started":"2022-07-24T01:55:22.567734Z","shell.execute_reply":"2022-07-24T01:55:22.575972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering\n1. Create FamilySize column from SibSp and Parch\n2. Create Title column from Name\n3. Encode all categorical columns except Embarked (has missing values)\n4. Drop unnecessary columns","metadata":{}},{"cell_type":"code","source":"# Crate FamilySize column from SibSp and Parch\nfor df in combine:\n    df['FamilySize'] = df['SibSp'] + df['Parch'] + 1","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:23.619308Z","iopub.execute_input":"2022-07-24T01:55:23.619732Z","iopub.status.idle":"2022-07-24T01:55:23.629986Z","shell.execute_reply.started":"2022-07-24T01:55:23.619697Z","shell.execute_reply":"2022-07-24T01:55:23.628494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Title column from Name\nfor df in combine:\n    df['Title'] = df['Name'].str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train['Title'], train['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:24.143667Z","iopub.execute_input":"2022-07-24T01:55:24.144433Z","iopub.status.idle":"2022-07-24T01:55:24.175776Z","shell.execute_reply.started":"2022-07-24T01:55:24.144397Z","shell.execute_reply":"2022-07-24T01:55:24.174175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replace titles to more common names and group rare titles\ncommon = ['Master', 'Mr', 'Miss', 'Mrs']\nfor df in combine:\n    df['Title'] = df['Title'].replace('Mlle', 'Miss')\n    df['Title'] = df['Title'].replace('Ms', 'Miss')\n    df['Title'] = df['Title'].replace('Mme', 'Mrs')\n    df['Title'] = [x if x in common else 'Rare' for x in df['Title']]\n\ntrain['Title'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:24.773707Z","iopub.execute_input":"2022-07-24T01:55:24.774417Z","iopub.status.idle":"2022-07-24T01:55:24.796002Z","shell.execute_reply.started":"2022-07-24T01:55:24.774379Z","shell.execute_reply":"2022-07-24T01:55:24.794614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode columns\nfor df in combine:\n    df['Sex'] = df['Sex'].map({'female': 0, 'male': 1})\n    #df['Title'] = df['Title'].map({'Mr': 0, 'Miss': 1, 'Mrs': 2, 'Master': 3, 'Rare': 4})\n    \ntitle_ohe1 = pd.get_dummies(train['Title'], prefix = 'Title', drop_first = True)\ntrain = pd.concat([train.drop('Title', axis = 1), title_ohe1], axis = 1)\n\ntitle_ohe2 = pd.get_dummies(test['Title'], prefix = 'Title', drop_first = True)\ntest = pd.concat([test.drop('Title', axis = 1), title_ohe2], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:25.713489Z","iopub.execute_input":"2022-07-24T01:55:25.714956Z","iopub.status.idle":"2022-07-24T01:55:25.735505Z","shell.execute_reply.started":"2022-07-24T01:55:25.714916Z","shell.execute_reply":"2022-07-24T01:55:25.734250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop Name, SibSp and Parch columns\ntrain = train.drop(['Name', 'SibSp', 'Parch'], axis = 1)\ntest = test.drop(['Name', 'SibSp', 'Parch'], axis = 1)\ncombine = [train, test]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:26.345269Z","iopub.execute_input":"2022-07-24T01:55:26.346041Z","iopub.status.idle":"2022-07-24T01:55:26.356632Z","shell.execute_reply.started":"2022-07-24T01:55:26.346003Z","shell.execute_reply":"2022-07-24T01:55:26.355691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Imputation : Age","metadata":{}},{"cell_type":"code","source":"# Check correlation of Age with other variables\nage_corr = train.corr().abs().unstack().sort_values(kind=\"quicksort\", ascending=False).reset_index()\nage_corr.rename(columns={\"level_0\": \"Feature 1\", \"level_1\": \"Feature 2\", 0: 'Correlation Coefficient'}, inplace=True)\nage_corr[age_corr['Feature 1'] == 'Age']","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:27.811832Z","iopub.execute_input":"2022-07-24T01:55:27.812584Z","iopub.status.idle":"2022-07-24T01:55:27.833922Z","shell.execute_reply.started":"2022-07-24T01:55:27.812530Z","shell.execute_reply":"2022-07-24T01:55:27.832374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Impute Age based on Pclass\nimpute_ages = np.zeros((2,3))\nfor df in combine:\n    for i in range(0, 2):\n        for j in range(0, 3):\n            impute_df = df[(df['Sex'] == i) & \\\n                                  (df['Pclass'] == j+1)]['Age'].dropna()\n            impute_ages[i,j] = int(impute_df.median())\n            \n    for i in range(0, 2):\n        for j in range(0, 3):\n            df.loc[ (df.Age.isnull()) & (df.Sex == i) & (df.Pclass == j+1), 'Age'] = impute_ages[i,j]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:29.025498Z","iopub.execute_input":"2022-07-24T01:55:29.026473Z","iopub.status.idle":"2022-07-24T01:55:29.071506Z","shell.execute_reply.started":"2022-07-24T01:55:29.026428Z","shell.execute_reply":"2022-07-24T01:55:29.070435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Imputation: Embarked","metadata":{}},{"cell_type":"code","source":"# Check correlation between Embarked & Pclass\nfig, ax=plt.subplots(1,figsize=(8,6))\nsns.countplot(x='Embarked',hue='Pclass', data=train)\nax.set_ylim(0,400)\nplt.title(\"Embarked vs Pclass\")\nprint('Check correlation between Embarked & Pclass\\n')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-24T01:55:31.074261Z","iopub.execute_input":"2022-07-24T01:55:31.074699Z","iopub.status.idle":"2022-07-24T01:55:31.349303Z","shell.execute_reply.started":"2022-07-24T01:55:31.074653Z","shell.execute_reply":"2022-07-24T01:55:31.347740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There seem to be some correlation between Embarked and Pclass as lower-class passengers were more likely to embarked from S and most passengers that embarked from C were from the upper-class","metadata":{}},{"cell_type":"code","source":"impute_embarked= ['', '', '']\nfor df in combine:\n    for i in range(0, 3):\n        impute_val = df[df['Pclass'] == i+1]['Embarked'].dropna().mode()[0]\n        impute_embarked[i] = impute_val\n        \n    for i in range(0, 3):\n        df.loc[ (df.Embarked.isnull()) & (df.Pclass == i+1), 'Embarked'] = impute_embarked[i]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:33.256470Z","iopub.execute_input":"2022-07-24T01:55:33.257409Z","iopub.status.idle":"2022-07-24T01:55:33.280732Z","shell.execute_reply.started":"2022-07-24T01:55:33.257357Z","shell.execute_reply":"2022-07-24T01:55:33.279583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for df in combine:\n#     df['Embarked'] = df['Embarked'].map({'S': 0, 'C': 1, 'Q': 2})\n\nembarked_ohe1 = pd.get_dummies(train['Embarked'], prefix = 'Embarked', drop_first = True)\ntrain = pd.concat([train.drop('Embarked', axis = 1), embarked_ohe1], axis = 1)\n\nembarked_ohe2 = pd.get_dummies(test['Embarked'], prefix = 'Embarked', drop_first = True)\ntest = pd.concat([test.drop('Embarked', axis = 1), embarked_ohe2], axis = 1)\n\ncombine = [train, test]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:34.674918Z","iopub.execute_input":"2022-07-24T01:55:34.675318Z","iopub.status.idle":"2022-07-24T01:55:34.688578Z","shell.execute_reply.started":"2022-07-24T01:55:34.675285Z","shell.execute_reply":"2022-07-24T01:55:34.687575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Imputation: Fare","metadata":{}},{"cell_type":"code","source":"imputer = SimpleImputer()\ntest['Fare'] = list(imputer.fit_transform(test[['Fare']]))\ntest['Fare'] = [x[0] for x in test['Fare']]\n\n# Check if there's any missing values left\nprint('Number of missing value(s) in every column (Train):')\nprint(train.isnull().sum())\nprint('Number of missing value(s) in every column (Test):')\nprint(test.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:55:36.358765Z","iopub.execute_input":"2022-07-24T01:55:36.359578Z","iopub.status.idle":"2022-07-24T01:55:36.380804Z","shell.execute_reply.started":"2022-07-24T01:55:36.359542Z","shell.execute_reply":"2022-07-24T01:55:36.379562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"code","source":"# Modeling\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost.sklearn import XGBClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.experimental import enable_hist_gradient_boosting\nfrom sklearn.ensemble import HistGradientBoostingClassifier\n\nfrom sklearn import metrics\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-24T01:52:32.865836Z","iopub.execute_input":"2022-07-24T01:52:32.867820Z","iopub.status.idle":"2022-07-24T01:52:34.413613Z","shell.execute_reply.started":"2022-07-24T01:52:32.867752Z","shell.execute_reply":"2022-07-24T01:52:34.412305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train.drop(columns = 'Survived')\nX = pd.get_dummies(X, drop_first = True)\n\ny = train['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:56:15.083601Z","iopub.execute_input":"2022-07-24T01:56:15.084020Z","iopub.status.idle":"2022-07-24T01:56:15.095760Z","shell.execute_reply.started":"2022-07-24T01:56:15.083987Z","shell.execute_reply":"2022-07-24T01:56:15.094624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:56:15.754213Z","iopub.execute_input":"2022-07-24T01:56:15.755348Z","iopub.status.idle":"2022-07-24T01:56:15.763017Z","shell.execute_reply.started":"2022-07-24T01:56:15.755305Z","shell.execute_reply":"2022-07-24T01:56:15.761919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_best_model(X_train, X_test, y_train, y_test):\n    # Logistic Regression\n    logreg = LogisticRegression(max_iter = 600, random_state = 42)\n    logreg.fit(X_train, y_train)\n    y_pred = logreg.predict(X_test)\n    logreg_acc = round(metrics.accuracy_score(y_test, y_pred) * 100, 2)\n    \n    # Decision Tree\n    decision_tree = DecisionTreeClassifier(random_state = 42)\n    decision_tree.fit(X_train, y_train)\n    y_pred = decision_tree.predict(X_test)\n    decision_tree_acc = round(metrics.accuracy_score(y_test, y_pred) * 100, 2)\n    \n    # Random Forest\n    random_forest = RandomForestClassifier(random_state = 42)\n    random_forest.fit(X_train, y_train)\n    y_pred = random_forest.predict(X_test)\n    random_forest_acc = round(metrics.accuracy_score(y_test, y_pred) * 100, 2)\n    \n    # XGBoost\n    xgb = XGBClassifier(random_state = 42)\n    xgb.fit(X_train, y_train)\n    y_pred = xgb.predict(X_test)\n    xgb_acc = round(metrics.accuracy_score(y_test, y_pred) * 100, 2)\n    \n    # GBM\n    gbm = GradientBoostingClassifier(random_state = 42)\n    gbm.fit(X_train, y_train)\n    y_pred = gbm.predict(X_test)\n    gbm_acc = round(metrics.accuracy_score(y_test, y_pred) * 100, 2)\n    \n    # LightGBM\n    lgbm = LGBMClassifier(random_state = 42)\n    lgbm.fit(X_train, y_train)\n    y_pred = lgbm.predict(X_test)\n    lgbm_acc = round(metrics.accuracy_score(y_test, y_pred) * 100, 2)\n        \n    # Catboost\n    catb = CatBoostClassifier(verbose = 0, random_state = 42)\n    catb.fit(X_train, y_train)\n    y_pred = catb.predict(X_test)\n    catb_acc = round(metrics.accuracy_score(y_test, y_pred) * 100, 2)\n    \n    # Histogram-based Gradient Boosting Classification Tree\n    hgb = HistGradientBoostingClassifier(random_state = 42)\n    hgb.fit(X_train, y_train)\n    y_pred = hgb.predict(X_test)\n    hgb_acc = round(metrics.accuracy_score(y_test, y_pred) * 100, 2)\n    \n    model_df = pd.DataFrame({'Model': ['Logistic Regression', 'Decision Tree', 'Random Forest', 'XGBoost', 'GBM', 'LightGBM', 'Catboost', 'HistBoost'],\n                       'Score': [logreg_acc, decision_tree_acc, random_forest_acc, xgb_acc, gbm_acc, lgbm_acc, catb_acc, hgb_acc]})\n    print(model_df.sort_values('Score', ascending = False).reset_index(drop = True))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:56:17.141051Z","iopub.execute_input":"2022-07-24T01:56:17.141816Z","iopub.status.idle":"2022-07-24T01:56:17.158886Z","shell.execute_reply.started":"2022-07-24T01:56:17.141759Z","shell.execute_reply":"2022-07-24T01:56:17.157873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"find_best_model(X_train, X_test, y_train, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T01:56:20.256747Z","iopub.execute_input":"2022-07-24T01:56:20.257280Z","iopub.status.idle":"2022-07-24T01:56:23.587606Z","shell.execute_reply.started":"2022-07-24T01:56:20.257243Z","shell.execute_reply":"2022-07-24T01:56:23.586167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our Random Forest model performed the best on unseen data.","metadata":{}},{"cell_type":"markdown","source":"# Random Forest\nFind the best hyperparameters for our Random Forest model using GridSearch and fit it accoring to those parameters.","metadata":{}},{"cell_type":"code","source":"rfc = RandomForestClassifier(random_state=42, \n                             n_jobs=-1 # Use all cores on your machine\n                            )\nparam_grid = { \n    'n_estimators': [100, 200, 300], # The number of boosting stages to perform\n    'max_features': ['auto'], # The number of features to consider when looking for the best split\n    'max_depth' : [4, 6, 8], # The maximum depth of the individual regression estimators.\n    'criterion' :['gini', 'entropy'] #Function to measure the quality of a split\n}\nCV_rfc = GridSearchCV(estimator=rfc, param_grid=param_grid, cv=5, scoring = 'accuracy', verbose = 10)\nCV_rfc.fit(X, y)\nprint('')\nprint('Best hyperparameters:',CV_rfc.best_params_)\n\nX_test = test.drop('PassengerId', axis = 1)\npredictions = CV_rfc.predict(X_test)\n\noutput = pd.DataFrame({'PassengerId': test.PassengerId,\n                      'Survived': predictions})\n\noutput.to_csv('titanic-submission.csv', index= False)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-24T01:57:33.653359Z","iopub.execute_input":"2022-07-24T01:57:33.654403Z","iopub.status.idle":"2022-07-24T01:58:25.332901Z","shell.execute_reply.started":"2022-07-24T01:57:33.654363Z","shell.execute_reply":"2022-07-24T01:58:25.331935Z"},"trusted":true},"execution_count":null,"outputs":[]}]}