{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T02:36:34.360550Z","iopub.execute_input":"2022-07-30T02:36:34.360995Z","iopub.status.idle":"2022-07-30T02:36:34.395631Z","shell.execute_reply.started":"2022-07-30T02:36:34.360911Z","shell.execute_reply":"2022-07-30T02:36:34.394555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import zscore\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\n\nfrom category_encoders import OneHotEncoder\nfrom sklearn.model_selection import train_test_split, cross_val_score, GridSearchCV\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import f1_score, make_scorer, accuracy_score\n\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom imblearn.over_sampling import RandomOverSampler\nfrom sklearn.impute import SimpleImputer","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:34.397523Z","iopub.execute_input":"2022-07-30T02:36:34.398471Z","iopub.status.idle":"2022-07-30T02:36:36.502710Z","shell.execute_reply.started":"2022-07-30T02:36:34.398431Z","shell.execute_reply":"2022-07-30T02:36:36.501422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/titanic/train.csv')\ndf_test = pd.read_csv('/kaggle/input/titanic/test.csv')\ndf_sample_sub = pd.read_csv('/kaggle/input/titanic/gender_submission.csv')\nid_test = df_test['PassengerId']\nprint(f'Training data shape is {df_train.shape}')\nprint(f'Testing data shape is {df_test.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:36.504800Z","iopub.execute_input":"2022-07-30T02:36:36.505703Z","iopub.status.idle":"2022-07-30T02:36:36.572367Z","shell.execute_reply.started":"2022-07-30T02:36:36.505654Z","shell.execute_reply":"2022-07-30T02:36:36.571071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:36.576575Z","iopub.execute_input":"2022-07-30T02:36:36.576952Z","iopub.status.idle":"2022-07-30T02:36:36.606018Z","shell.execute_reply.started":"2022-07-30T02:36:36.576917Z","shell.execute_reply":"2022-07-30T02:36:36.604713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:36.607573Z","iopub.execute_input":"2022-07-30T02:36:36.607969Z","iopub.status.idle":"2022-07-30T02:36:36.629313Z","shell.execute_reply.started":"2022-07-30T02:36:36.607938Z","shell.execute_reply":"2022-07-30T02:36:36.628045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:36.631270Z","iopub.execute_input":"2022-07-30T02:36:36.631992Z","iopub.status.idle":"2022-07-30T02:36:36.656746Z","shell.execute_reply.started":"2022-07-30T02:36:36.631946Z","shell.execute_reply":"2022-07-30T02:36:36.655589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Describing the data","metadata":{}},{"cell_type":"code","source":"df_train.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:36.658299Z","iopub.execute_input":"2022-07-30T02:36:36.659463Z","iopub.status.idle":"2022-07-30T02:36:36.712522Z","shell.execute_reply.started":"2022-07-30T02:36:36.659416Z","shell.execute_reply":"2022-07-30T02:36:36.711439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, only the features ```Cabin``` and ```Age``` have null values. Since the Cabin has more than 50% null values, we delete that later, and impute Age null values.","metadata":{}},{"cell_type":"markdown","source":"# Exploratory data analysis","metadata":{}},{"cell_type":"markdown","source":"**Let's explore first numerical values**","metadata":{}},{"cell_type":"code","source":"df_train['Survived'].replace({1:\"Survived\", 0: \"Didn't Survived\"}).value_counts(normalize=True).plot(kind='barh');","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:36.714621Z","iopub.execute_input":"2022-07-30T02:36:36.715071Z","iopub.status.idle":"2022-07-30T02:36:36.928885Z","shell.execute_reply.started":"2022-07-30T02:36:36.715026Z","shell.execute_reply":"2022-07-30T02:36:36.927589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that around 40% is the survival rate of the passengers","metadata":{}},{"cell_type":"code","source":"sns.countplot(data = df_train, x = 'Sex', hue='Survived');","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:36.933785Z","iopub.execute_input":"2022-07-30T02:36:36.936312Z","iopub.status.idle":"2022-07-30T02:36:37.132431Z","shell.execute_reply.started":"2022-07-30T02:36:36.936267Z","shell.execute_reply":"2022-07-30T02:36:37.130942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- majority of the passengers are male\n- female has higher survival rate than male","metadata":{}},{"cell_type":"markdown","source":"Now, Let's look the distribution of age in terms of the target","metadata":{}},{"cell_type":"code","source":"sns.histplot(data = df_train, x = 'Age', hue='Survived', stat='percent');","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:37.134016Z","iopub.execute_input":"2022-07-30T02:36:37.134511Z","iopub.status.idle":"2022-07-30T02:36:37.493649Z","shell.execute_reply.started":"2022-07-30T02:36:37.134427Z","shell.execute_reply":"2022-07-30T02:36:37.492023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Since most passengers are between 20 - 50 years old, survival rate in this age range is not high\n- Age with <10 has a high survival rate\n- the oldest survived","metadata":{}},{"cell_type":"markdown","source":"Now, Let's look the distribution of pclass in terms of the target","metadata":{}},{"cell_type":"code","source":"sns.countplot(data = df_train, x = 'Pclass', hue='Survived');","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:37.495452Z","iopub.execute_input":"2022-07-30T02:36:37.496140Z","iopub.status.idle":"2022-07-30T02:36:37.714898Z","shell.execute_reply.started":"2022-07-30T02:36:37.496098Z","shell.execute_reply":"2022-07-30T02:36:37.714036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- majority of passengers were in class 3\n- class 3 had the lowest survival rate\n- class 2 had lower survival rate\n- class 1 had the highest survival rate. This makes sense since the people in class 1 are considered rich people","metadata":{}},{"cell_type":"markdown","source":"Now let's observe sibsp, and Parch to see if the survival rate of people who are together with spouse/siblings/parents/childeren is high","metadata":{}},{"cell_type":"code","source":"col = ['SibSp', 'Parch']\n\nfor i in col:\n    sns.countplot(data = df_train, x = i, hue='Survived')\n    plt.show();","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:37.716192Z","iopub.execute_input":"2022-07-30T02:36:37.717056Z","iopub.status.idle":"2022-07-30T02:36:38.270650Z","shell.execute_reply.started":"2022-07-30T02:36:37.717019Z","shell.execute_reply":"2022-07-30T02:36:38.269464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- majority of the passenger do not have siblings/spouse or accompanied by their nannies\n- the survival rate of those with only 1 sibling/children is high\n- children without parents abourd and adults without spouse or siblings have the lowest survival rate\n- people with most relatives aboard died\n- we can transform this to the number of relatives by adding the two features","metadata":{}},{"cell_type":"markdown","source":"Now, let's check the correlation of the class and the fare. We delete one if they are correlated","metadata":{}},{"cell_type":"code","source":"sns.catplot(data=df_train, x=\"Pclass\", y=\"Fare\", hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:38.271959Z","iopub.execute_input":"2022-07-30T02:36:38.272311Z","iopub.status.idle":"2022-07-30T02:36:38.783036Z","shell.execute_reply.started":"2022-07-30T02:36:38.272278Z","shell.execute_reply":"2022-07-30T02:36:38.782153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- class 1 has higher fare and high survival rate\n- class 2 and 3 fare do not have significant difference and low survival rate\n- There is moderate correlation, we are going to delete the fare on our feature selection section","metadata":{}},{"cell_type":"markdown","source":"Lastly, let's check **categorical features**. Let's do the following:\n- Remove features with more than 50% null values\n- Remove features with high/low cardinalty\n- check the distribution","metadata":{}},{"cell_type":"code","source":"df_train.select_dtypes('object').nunique() / len(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:38.784092Z","iopub.execute_input":"2022-07-30T02:36:38.784847Z","iopub.status.idle":"2022-07-30T02:36:38.798248Z","shell.execute_reply.started":"2022-07-30T02:36:38.784810Z","shell.execute_reply":"2022-07-30T02:36:38.796930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Name and Ticket are high-cardinality values","metadata":{}},{"cell_type":"code","source":"df_train.select_dtypes('object').isnull().sum() / len(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:38.799923Z","iopub.execute_input":"2022-07-30T02:36:38.800384Z","iopub.status.idle":"2022-07-30T02:36:38.814471Z","shell.execute_reply.started":"2022-07-30T02:36:38.800350Z","shell.execute_reply":"2022-07-30T02:36:38.813472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Cabin has more than 50% null values","metadata":{}},{"cell_type":"markdown","source":"There is only one categorical feature, let's look at the relationship embarked to the target vector using chi2","metadata":{}},{"cell_type":"code","source":"from scipy.stats import chi2_contingency\nfrom scipy.stats import chi2\ndf_new = df_train[['Embarked', 'Survived']]\ndf_new = df_new.fillna(df_new['Embarked'].mode()[0]) #replace null values with most frequent\ndf_new['Embarked'] = df_new['Embarked'].replace({'C': 1, 'S': 2, 'Q': 3})","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:38.816094Z","iopub.execute_input":"2022-07-30T02:36:38.817832Z","iopub.status.idle":"2022-07-30T02:36:38.832330Z","shell.execute_reply.started":"2022-07-30T02:36:38.817783Z","shell.execute_reply":"2022-07-30T02:36:38.831169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# contingency table\nstat, p, dof, expected = chi2_contingency(df_new)\n# interpret test-statistic\nprob = 0.95\ncritical = chi2.ppf(prob, dof)\n#Hypotheses\nprint('H0: There is no significant relationship between the embarked and the survived')\nprint('H1: There is a significant relationship between the embarked and the survived')\n#using CV method\nif abs(stat) >= critical:\n    print('Dependent (reject H0)')\nelse:\n    print('Independent (fail to reject H0)')\n# using p-value method\nalpha = 0.05 - prob\nif p <= alpha:\n    print('Dependent (reject H0)')\nelse:\n    print('Independent (fail to reject H0)')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:38.834074Z","iopub.execute_input":"2022-07-30T02:36:38.835368Z","iopub.status.idle":"2022-07-30T02:36:38.850584Z","shell.execute_reply.started":"2022-07-30T02:36:38.834754Z","shell.execute_reply":"2022-07-30T02:36:38.848883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we can see that the two variables are not associated with each other. We can remove the feature Embarked","metadata":{}},{"cell_type":"markdown","source":"**Conclusion**\n- All categorical features will be removed except ```Sex```\n- Age, Pclass, and Gender are major factors to survive\n- We can Tansform Parch and sibsp to the ```number of relatives``` by adding the two features will try this in the next revision\n- Class and Fare are multicollinearities","metadata":{}},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"markdown","source":"## Feature Selection","metadata":{}},{"cell_type":"code","source":"def wrangle(file_path):\n    # Read csv file\n    df = pd.read_csv(file_path)\n    # Remove values with more than 50% null values\n    df.drop(columns = 'Cabin', inplace = True)\n    # Remove high-cardinality features\n    df.drop(columns = ['Name', 'PassengerId'], inplace = True)\n    # Remove Multicollinearities\n    df.drop(columns = 'Fare', inplace = True)\n    # Remove all categorical features\n    cat_cols = df.select_dtypes('object').drop(columns='Sex').columns.tolist()\n    df.drop(columns = cat_cols, inplace = True)\n    # Replace null values with median\n    df['Age'] = df['Age'].fillna(df['Age'].median())\n    # Replace 'Sex' to numerical\n    df['Sex'] = df['Sex'].replace({'female': 0, 'male': 1})\n    return df\n\ndf = wrangle('/kaggle/input/titanic/train.csv')\ndf_test = wrangle('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:38.853020Z","iopub.execute_input":"2022-07-30T02:36:38.854184Z","iopub.status.idle":"2022-07-30T02:36:38.896965Z","shell.execute_reply.started":"2022-07-30T02:36:38.854131Z","shell.execute_reply":"2022-07-30T02:36:38.895691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:38.898970Z","iopub.execute_input":"2022-07-30T02:36:38.899393Z","iopub.status.idle":"2022-07-30T02:36:38.913798Z","shell.execute_reply.started":"2022-07-30T02:36:38.899350Z","shell.execute_reply":"2022-07-30T02:36:38.912548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**This is the final data**","metadata":{}},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"Let's check the distribution of the data","metadata":{}},{"cell_type":"code","source":"df.drop(columns='Survived').hist(figsize=(20,15), bins = 10);","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:38.915456Z","iopub.execute_input":"2022-07-30T02:36:38.916479Z","iopub.status.idle":"2022-07-30T02:36:39.909914Z","shell.execute_reply.started":"2022-07-30T02:36:38.916433Z","shell.execute_reply":"2022-07-30T02:36:39.908705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that the sibsp and Parch is highly skewed, we can transform that to log","metadata":{}},{"cell_type":"code","source":"cols = ['Parch', 'SibSp']\nfor col in cols:\n    df[col] = np.log(df[col] + 1)\n    df_test[col] = np.log(df_test[col] + 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:39.911668Z","iopub.execute_input":"2022-07-30T02:36:39.912079Z","iopub.status.idle":"2022-07-30T02:36:39.922417Z","shell.execute_reply.started":"2022-07-30T02:36:39.912045Z","shell.execute_reply":"2022-07-30T02:36:39.921030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's scale the age using standardscaler","metadata":{}},{"cell_type":"code","source":"ss = StandardScaler()\nage =  np.asarray(df['Age'])\nage = age.reshape(-1, 1)\ndf['Age'] = ss.fit_transform(age)\nage_test =  np.asarray(df_test['Age'])\nage_test = age_test.reshape(-1, 1)\ndf_test['Age'] = ss.fit_transform(age_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:39.923584Z","iopub.execute_input":"2022-07-30T02:36:39.924399Z","iopub.status.idle":"2022-07-30T02:36:39.938697Z","shell.execute_reply.started":"2022-07-30T02:36:39.924361Z","shell.execute_reply":"2022-07-30T02:36:39.937814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:39.939968Z","iopub.execute_input":"2022-07-30T02:36:39.940527Z","iopub.status.idle":"2022-07-30T02:36:39.954996Z","shell.execute_reply.started":"2022-07-30T02:36:39.940473Z","shell.execute_reply":"2022-07-30T02:36:39.953595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split","metadata":{}},{"cell_type":"code","source":"target = 'Survived'\nX = df.drop(columns = target)\ny = df[target]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:39.960812Z","iopub.execute_input":"2022-07-30T02:36:39.961210Z","iopub.status.idle":"2022-07-30T02:36:39.968725Z","shell.execute_reply.started":"2022-07-30T02:36:39.961167Z","shell.execute_reply":"2022-07-30T02:36:39.967697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size = 0.20, random_state = 42)\nprint(f'Training Data Shape: {X_train.shape} , {y_train.shape}')\nprint(f'Validation Data Shape: {X_val.shape} , {y_val.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:39.969984Z","iopub.execute_input":"2022-07-30T02:36:39.970888Z","iopub.status.idle":"2022-07-30T02:36:39.982108Z","shell.execute_reply.started":"2022-07-30T02:36:39.970852Z","shell.execute_reply":"2022-07-30T02:36:39.980866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Oversampling","metadata":{}},{"cell_type":"code","source":"y_train.value_counts().plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:39.986281Z","iopub.execute_input":"2022-07-30T02:36:39.987232Z","iopub.status.idle":"2022-07-30T02:36:40.143481Z","shell.execute_reply.started":"2022-07-30T02:36:39.987163Z","shell.execute_reply":"2022-07-30T02:36:40.142178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"as we can see, our data is imbalance. Let's try oversampling to balance the data","metadata":{}},{"cell_type":"code","source":"over_sampler = RandomOverSampler()\nX_train_over, y_train_over = over_sampler.fit_resample(X_train, y_train)\nprint(f'Feature Matrix shape{X_train.shape}')\nprint(f'Target Vector shape{y_train.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:45:24.064699Z","iopub.execute_input":"2022-07-30T02:45:24.065064Z","iopub.status.idle":"2022-07-30T02:45:24.079063Z","shell.execute_reply.started":"2022-07-30T02:45:24.065036Z","shell.execute_reply":"2022-07-30T02:45:24.077864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_scores = []\naccuracy_scores = []\n\ndef fit_model(model, X_train, y_train):\n#     model = make_pipeline(OneHotEncoder(), clf)\n    model.fit(X_train, y_train)\n    f1 = f1_score(y_val, model.predict(X_val))\n    accuracy = accuracy_score(y_val, model.predict(X_val))\n    f1_scores.append(f1)\n    accuracy_scores.append(accuracy)\n\n    print(f'F1 Scores: {f1}')\n    print(f'Accuracy Scores: {accuracy}')\n    print(f'Cross Validation Scores: {cross_val_score(model, X_train, y_train, cv = 5)}')\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:48:25.863948Z","iopub.execute_input":"2022-07-30T02:48:25.864338Z","iopub.status.idle":"2022-07-30T02:48:25.872268Z","shell.execute_reply.started":"2022-07-30T02:48:25.864305Z","shell.execute_reply":"2022-07-30T02:48:25.871049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg = fit_model(LogisticRegression(), X_train_over, y_train_over )\nDT = fit_model(DecisionTreeClassifier(random_state = 42), X_train_over, y_train_over )\nRF = fit_model(RandomForestClassifier(random_state = 42), X_train_over, y_train_over )\nGB = fit_model(GradientBoostingClassifier(random_state = 42), X_train_over, y_train_over )","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:48:49.898212Z","iopub.execute_input":"2022-07-30T02:48:49.898605Z","iopub.status.idle":"2022-07-30T02:48:51.937585Z","shell.execute_reply.started":"2022-07-30T02:48:49.898574Z","shell.execute_reply":"2022-07-30T02:48:51.936439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyper Parameter Tuning","metadata":{"execution":{"iopub.status.busy":"2022-07-26T10:04:19.815442Z","iopub.execute_input":"2022-07-26T10:04:19.815915Z","iopub.status.idle":"2022-07-26T10:04:19.821378Z","shell.execute_reply.started":"2022-07-26T10:04:19.815876Z","shell.execute_reply":"2022-07-26T10:04:19.820332Z"}}},{"cell_type":"code","source":"def tune_params(clf, params = None):\n    import time\n#     model = make_pipeline(OneHotEncoder(handle_unknown='error'), clf)\n    tuned_model = GridSearchCV(\n        clf,\n        param_grid=params,\n        scoring=['f1', 'accuracy'],\n        cv =5,\n        refit='accuracy',\n        n_jobs = -1,\n        verbose = 1\n    )\n    starting_time = time.time()\n    tuned_model.fit(X_train_over, y_train_over)\n    ending_time = time.time()\n    elapse_time = ending_time - starting_time\n    return tuned_model, elapse_time","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:54:59.542574Z","iopub.execute_input":"2022-07-30T02:54:59.542957Z","iopub.status.idle":"2022-07-30T02:54:59.550534Z","shell.execute_reply.started":"2022-07-30T02:54:59.542926Z","shell.execute_reply":"2022-07-30T02:54:59.549341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt_params = {\n    'splitter': ['best', 'random'],\n    'max_depth': range(1, 31, 5),\n}\n\nensemble_params = {\n    'n_estimators': range(10, 150, 10),\n    'max_depth': range(1, 31, 5),\n}\n\nDT_final_model, DT_training_time = tune_params(DecisionTreeClassifier(random_state = 42), dt_params)\nRF_final_model, RF_training_time = tune_params(RandomForestClassifier(random_state = 42), ensemble_params)\nGB_final_model, GB_training_time = tune_params(GradientBoostingClassifier(random_state = 42), ensemble_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:36:42.292281Z","iopub.execute_input":"2022-07-30T02:36:42.292647Z","iopub.status.idle":"2022-07-30T02:38:07.526777Z","shell.execute_reply.started":"2022-07-30T02:36:42.292616Z","shell.execute_reply":"2022-07-30T02:38:07.525576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(DT_final_model.cv_results_).sort_values(by='rank_test_accuracy').head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:38:07.528768Z","iopub.execute_input":"2022-07-30T02:38:07.529217Z","iopub.status.idle":"2022-07-30T02:38:07.564711Z","shell.execute_reply.started":"2022-07-30T02:38:07.529170Z","shell.execute_reply":"2022-07-30T02:38:07.563638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(RF_final_model.cv_results_).sort_values(by='rank_test_accuracy').head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:38:07.566291Z","iopub.execute_input":"2022-07-30T02:38:07.566976Z","iopub.status.idle":"2022-07-30T02:38:07.600461Z","shell.execute_reply.started":"2022-07-30T02:38:07.566940Z","shell.execute_reply":"2022-07-30T02:38:07.599336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(GB_final_model.cv_results_).sort_values(by='rank_test_accuracy').head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:38:07.601976Z","iopub.execute_input":"2022-07-30T02:38:07.602268Z","iopub.status.idle":"2022-07-30T02:38:07.633451Z","shell.execute_reply.started":"2022-07-30T02:38:07.602242Z","shell.execute_reply":"2022-07-30T02:38:07.632573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(DT_final_model.best_score_)\nprint(RF_final_model.best_score_)\nprint(GB_final_model.best_score_)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:38:07.634561Z","iopub.execute_input":"2022-07-30T02:38:07.635186Z","iopub.status.idle":"2022-07-30T02:38:07.641727Z","shell.execute_reply.started":"2022-07-30T02:38:07.635142Z","shell.execute_reply":"2022-07-30T02:38:07.640164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(DT_final_model.best_params_)\nprint(RF_final_model.best_params_)\nprint(GB_final_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:38:07.642902Z","iopub.execute_input":"2022-07-30T02:38:07.643741Z","iopub.status.idle":"2022-07-30T02:38:07.654454Z","shell.execute_reply.started":"2022-07-30T02:38:07.643706Z","shell.execute_reply":"2022-07-30T02:38:07.653357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since we already know the parameters needed, let's retrain the data without split then do another oversampling\n","metadata":{}},{"cell_type":"code","source":"over_sampler = RandomOverSampler()\nX_over, y_over = over_sampler.fit_resample(X, y)\nprint(f'Feature Matrix shape{X_over.shape}')\nprint(f'Target Vector shape{y_over.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:43:15.319180Z","iopub.execute_input":"2022-07-30T02:43:15.319586Z","iopub.status.idle":"2022-07-30T02:43:15.334173Z","shell.execute_reply.started":"2022-07-30T02:43:15.319553Z","shell.execute_reply":"2022-07-30T02:43:15.332851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Retrain the Model","metadata":{}},{"cell_type":"code","source":"log_reg = fit_model(LogisticRegression(), X_over, y_over)\nDT = fit_model(DecisionTreeClassifier(max_depth = 11, splitter='best', random_state = 42), X_over, y_over)\nRF = fit_model(RandomForestClassifier(max_depth = 11, n_estimators = 90, random_state = 42), X_over, y_over)\nGB = fit_model(GradientBoostingClassifier(max_depth = 11, n_estimators = 50, random_state = 42), X_over, y_over)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:50:52.009296Z","iopub.execute_input":"2022-07-30T02:50:52.010234Z","iopub.status.idle":"2022-07-30T02:50:55.255628Z","shell.execute_reply.started":"2022-07-30T02:50:52.010183Z","shell.execute_reply":"2022-07-30T02:50:55.253889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The best algorithm is GradientBoosting","metadata":{}},{"cell_type":"code","source":"y_pred = GB.predict(df_test)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:51:15.969265Z","iopub.execute_input":"2022-07-30T02:51:15.969684Z","iopub.status.idle":"2022-07-30T02:51:15.983142Z","shell.execute_reply.started":"2022-07-30T02:51:15.969649Z","shell.execute_reply":"2022-07-30T02:51:15.981742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.DataFrame({\n    'PassengerId': id_test,\n    'Survived': y_pred\n})\nsubmit.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:51:17.991287Z","iopub.execute_input":"2022-07-30T02:51:17.991704Z","iopub.status.idle":"2022-07-30T02:51:18.002450Z","shell.execute_reply.started":"2022-07-30T02:51:17.991666Z","shell.execute_reply":"2022-07-30T02:51:18.001291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('Submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T02:38:10.454374Z","iopub.execute_input":"2022-07-30T02:38:10.455268Z","iopub.status.idle":"2022-07-30T02:38:10.468190Z","shell.execute_reply.started":"2022-07-30T02:38:10.455227Z","shell.execute_reply":"2022-07-30T02:38:10.466778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}