{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.feature_selection import mutual_info_classif\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import (\n    OrdinalEncoder,\n    OneHotEncoder,\n    StandardScaler,\n    label_binarize\n)\nfrom sklearn.model_selection import train_test_split, learning_curve\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import (\n    confusion_matrix,\n    ConfusionMatrixDisplay,\n    roc_auc_score,\n    roc_curve,\n    auc,\n    RocCurveDisplay,\n    accuracy_score\n)\n\nfrom xgboost import XGBClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:21.703511Z","iopub.execute_input":"2022-07-17T15:10:21.703879Z","iopub.status.idle":"2022-07-17T15:10:22.640679Z","shell.execute_reply.started":"2022-07-17T15:10:21.703853Z","shell.execute_reply":"2022-07-17T15:10:22.639549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data analysis","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/titanic/train.csv\")\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.643837Z","iopub.execute_input":"2022-07-17T15:10:22.645084Z","iopub.status.idle":"2022-07-17T15:10:22.685461Z","shell.execute_reply.started":"2022-07-17T15:10:22.645046Z","shell.execute_reply":"2022-07-17T15:10:22.684092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop(columns=[\"PassengerId\", \"Name\", \"Cabin\", \"Ticket\"])\nX.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.686869Z","iopub.execute_input":"2022-07-17T15:10:22.687525Z","iopub.status.idle":"2022-07-17T15:10:22.709963Z","shell.execute_reply.started":"2022-07-17T15:10:22.687495Z","shell.execute_reply":"2022-07-17T15:10:22.709098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There are only 2 missing values in embarked,\n# so it is no matter how to impute them\nX.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.712270Z","iopub.execute_input":"2022-07-17T15:10:22.712795Z","iopub.status.idle":"2022-07-17T15:10:22.728654Z","shell.execute_reply.started":"2022-07-17T15:10:22.712765Z","shell.execute_reply":"2022-07-17T15:10:22.726792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill categorical feature by most frequent\nX[\"Embarked\"].fillna(X[\"Embarked\"].mode()[0], inplace=True)\n# fill age feature by median\nX[\"Age\"].fillna(X[\"Age\"].mean(), inplace=True)\nX.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.730412Z","iopub.execute_input":"2022-07-17T15:10:22.731300Z","iopub.status.idle":"2022-07-17T15:10:22.749443Z","shell.execute_reply.started":"2022-07-17T15:10:22.731262Z","shell.execute_reply":"2022-07-17T15:10:22.748334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# object_cols = X.select_dtypes(include=\"object\").columns\n# X = X.drop(object_cols, axis=1).join(pd.get_dummies(X[object_cols]))\n# X.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.751321Z","iopub.execute_input":"2022-07-17T15:10:22.751719Z","iopub.status.idle":"2022-07-17T15:10:22.756699Z","shell.execute_reply.started":"2022-07-17T15:10:22.751684Z","shell.execute_reply":"2022-07-17T15:10:22.755508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pd_ordinal_encoder(df):\n    \"\"\"\n    Function to encode categorical columns in\n    pd.DataFrame `df` ordinally using only pandas.\n    \"\"\"\n    object_cols = df.select_dtypes(include=\"object\").columns\n    for col in object_cols:\n        ordinal_dict = {val: i for i, val in enumerate(df[col].unique())}\n        df[col].replace(ordinal_dict, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.758139Z","iopub.execute_input":"2022-07-17T15:10:22.758805Z","iopub.status.idle":"2022-07-17T15:10:22.771220Z","shell.execute_reply.started":"2022-07-17T15:10:22.758772Z","shell.execute_reply":"2022-07-17T15:10:22.770054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd_ordinal_encoder(X)\nX.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.772256Z","iopub.execute_input":"2022-07-17T15:10:22.772691Z","iopub.status.idle":"2022-07-17T15:10:22.797481Z","shell.execute_reply.started":"2022-07-17T15:10:22.772667Z","shell.execute_reply":"2022-07-17T15:10:22.796611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = X.pop(\"Survived\")\ndiscrete_features = X.dtypes == int","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.798864Z","iopub.execute_input":"2022-07-17T15:10:22.799654Z","iopub.status.idle":"2022-07-17T15:10:22.805956Z","shell.execute_reply.started":"2022-07-17T15:10:22.799618Z","shell.execute_reply":"2022-07-17T15:10:22.804723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mi_scores = mutual_info_classif(X, y, discrete_features=discrete_features)\nmi_scores = pd.Series(mi_scores, name=\"MI scores\", index=X.columns)\nmi_scores = mi_scores.sort_values(ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.809804Z","iopub.execute_input":"2022-07-17T15:10:22.810997Z","iopub.status.idle":"2022-07-17T15:10:22.842599Z","shell.execute_reply.started":"2022-07-17T15:10:22.810940Z","shell.execute_reply":"2022-07-17T15:10:22.841831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(dpi=100, figsize=(8, 5))\nwidth = np.arange(len(mi_scores))\nticks = list(mi_scores.index)\nplt.barh(width, mi_scores)\nplt.yticks(width, ticks)\nplt.title(\"Mutual information scores\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:22.843632Z","iopub.execute_input":"2022-07-17T15:10:22.844943Z","iopub.status.idle":"2022-07-17T15:10:23.036337Z","shell.execute_reply.started":"2022-07-17T15:10:22.844903Z","shell.execute_reply":"2022-07-17T15:10:23.035588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.groupby(\"Sex\", as_index=False)[\"Survived\"].mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.037316Z","iopub.execute_input":"2022-07-17T15:10:23.038157Z","iopub.status.idle":"2022-07-17T15:10:23.052458Z","shell.execute_reply.started":"2022-07-17T15:10:23.038130Z","shell.execute_reply":"2022-07-17T15:10:23.051616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.groupby(\"Pclass\", as_index=False)[\"Survived\"].mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.054801Z","iopub.execute_input":"2022-07-17T15:10:23.055517Z","iopub.status.idle":"2022-07-17T15:10:23.068154Z","shell.execute_reply.started":"2022-07-17T15:10:23.055474Z","shell.execute_reply":"2022-07-17T15:10:23.066769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = sns.FacetGrid(df_train, col=\"Survived\")\ng.map_dataframe(sns.histplot, x=\"Age\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.069688Z","iopub.execute_input":"2022-07-17T15:10:23.069989Z","iopub.status.idle":"2022-07-17T15:10:23.434804Z","shell.execute_reply.started":"2022-07-17T15:10:23.069965Z","shell.execute_reply":"2022-07-17T15:10:23.433641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = sns.boxplot(x=\"Survived\", y=\"Fare\", data=df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.436365Z","iopub.execute_input":"2022-07-17T15:10:23.436731Z","iopub.status.idle":"2022-07-17T15:10:23.567383Z","shell.execute_reply.started":"2022-07-17T15:10:23.436695Z","shell.execute_reply":"2022-07-17T15:10:23.566321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_no_outliers = df_train.drop(df_train[df_train[\"Fare\"] > 200].index)\ng = sns.boxplot(x=\"Survived\", y=\"Fare\", data=df_train_no_outliers)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.568878Z","iopub.execute_input":"2022-07-17T15:10:23.569224Z","iopub.status.idle":"2022-07-17T15:10:23.721813Z","shell.execute_reply.started":"2022-07-17T15:10:23.569189Z","shell.execute_reply":"2022-07-17T15:10:23.720589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There are some weak outliers in age and `mean`\n# imputing strategy should be good choice\ng = sns.boxplot(x=\"Sex\", y=\"Age\", data=df_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.723774Z","iopub.execute_input":"2022-07-17T15:10:23.724400Z","iopub.status.idle":"2022-07-17T15:10:23.885247Z","shell.execute_reply.started":"2022-07-17T15:10:23.724361Z","shell.execute_reply":"2022-07-17T15:10:23.884398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building pipeline","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/titanic/train.csv\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.886667Z","iopub.execute_input":"2022-07-17T15:10:23.887140Z","iopub.status.idle":"2022-07-17T15:10:23.908905Z","shell.execute_reply.started":"2022-07-17T15:10:23.887107Z","shell.execute_reply":"2022-07-17T15:10:23.907851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop(columns=[\"PassengerId\", \"Name\", \"Ticket\", \"Cabin\"])\ny = X.pop(\"Survived\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.913596Z","iopub.execute_input":"2022-07-17T15:10:23.913884Z","iopub.status.idle":"2022-07-17T15:10:23.920329Z","shell.execute_reply.started":"2022-07-17T15:10:23.913858Z","shell.execute_reply":"2022-07-17T15:10:23.919505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y, train_size=0.75, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.922049Z","iopub.execute_input":"2022-07-17T15:10:23.922504Z","iopub.status.idle":"2022-07-17T15:10:23.934763Z","shell.execute_reply.started":"2022-07-17T15:10:23.922474Z","shell.execute_reply":"2022-07-17T15:10:23.933886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_learning_curve(\n    estimator,\n    title,\n    X,\n    y,\n    axes=None,\n    ylim=None,\n    cv=None,\n    n_jobs=None,\n    train_sizes=np.linspace(0.1, 1.0, 10),\n):\n    \"\"\"\n    Modified plot_learning_curve function code\n    from `https://scikit-learn.org/stable/auto_examples/model_selection/plot_learning_curve.html`.\n    Using nanmean and nanstd instead of mean and std to avoid\n    miss of points from plots.\n    \"\"\"\n    if axes is None:\n        _, axes = plt.subplots(1, 3, figsize=(20, 5))\n        \n    axes[0].set_title(title)\n    if ylim is not None:\n        axes[0].set_ylim(*ylim)\n    axes[0].set_xlabel(\"Training examples\")\n    axes[0].set_ylabel(\"Score\")\n    \n    train_sizes, train_scores, test_scores, fit_times, _ = learning_curve(\n        estimator,\n        X,\n        y,\n        cv=cv,\n        n_jobs=n_jobs,\n        train_sizes=train_sizes,\n        return_times=True,\n    )\n    \n    train_scores_mean = np.nanmean(train_scores, axis=1)\n    train_scores_std = np.nanstd(train_scores, axis=1)\n    test_scores_mean = np.nanmean(test_scores, axis=1)\n    test_scores_std = np.nanstd(test_scores, axis=1)\n    fit_times_mean = np.nanmean(fit_times, axis=1)\n    fit_times_std = np.nanstd(fit_times, axis=1)\n    \n    axes[0].grid()\n    axes[0].fill_between(\n        train_sizes,\n        train_scores_mean - train_scores_std,\n        train_scores_mean + train_scores_std,\n        alpha = 0.1,\n        color = \"r\"\n    )\n    axes[0].fill_between(\n        train_sizes,\n        test_scores_mean - test_scores_std,\n        test_scores_mean + test_scores_std,\n        alpha = 0.1,\n        color = \"g\"\n    )\n    axes[0].plot(\n        train_sizes, train_scores_mean, \"o-\", color=\"r\", label=\"Training score\"\n    )\n    axes[0].plot(\n        train_sizes, test_scores_mean, \"o-\", color=\"g\", label=\"Cross-validation score\"\n    )\n    axes[0].legend(loc=\"best\")\n    \n    axes[1].grid()\n    axes[1].plot(train_sizes, fit_times_mean, \"o-\")\n    axes[1].fill_between(\n        train_sizes,\n        fit_times_mean - fit_times_std,\n        fit_times_mean + fit_times_std,\n        alpha = 0.1\n    )\n    axes[1].set_xlabel(\"Training examples\")\n    axes[1].set_ylabel(\"fit_times\")\n    axes[1].set_title(\"Scalability of the model\")\n    \n    fit_time_argsort = fit_times_mean.argsort()\n    fit_time_sorted = fit_times_mean[fit_time_argsort]\n    test_scores_mean_sorted = test_scores_mean[fit_time_argsort]\n    test_scores_std_sorted = test_scores_std[fit_time_argsort]\n    axes[2].grid()\n    axes[2].plot(fit_time_sorted, test_scores_mean_sorted, \"o-\")\n    axes[2].fill_between(\n        fit_time_sorted,\n        test_scores_mean_sorted - test_scores_std_sorted,\n        test_scores_mean_sorted + test_scores_std_sorted,\n        alpha = 0.1\n    )\n    axes[2].set_xlabel(\"fit_times\")\n    axes[2].set_ylabel(\"Score\")\n    axes[2].set_title(\"Performance of the model\")\n    \n    return axes","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.936079Z","iopub.execute_input":"2022-07-17T15:10:23.936527Z","iopub.status.idle":"2022-07-17T15:10:23.959547Z","shell.execute_reply.started":"2022-07-17T15:10:23.936499Z","shell.execute_reply":"2022-07-17T15:10:23.958667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def roc_plot(y_valid, preds):\n    classes = np.unique(y_valid.values)\n    y_test = label_binarize(y_valid, classes=classes)\n    y_pred = label_binarize(preds, classes=classes)\n    n_classes = classes.size\n    \n    fpr, tpr, _ = roc_curve(y_test, y_pred)\n    roc_auc = auc(fpr, tpr)\n    \n    plt.figure()\n    lw = 2\n    plt.plot(\n        fpr,\n        tpr,\n        color=\"darkorange\",\n        lw=lw,\n        label=\"ROC curve (area = %0.2f)\" % roc_auc,\n    )\n    plt.plot([0, 1], [0, 1], color=\"navy\", lw=lw, linestyle=\"--\")\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel(\"False Positive Rate\")\n    plt.ylabel(\"True Positive Rate\")\n    plt.title(\"ROC curve\")\n    plt.legend(loc=\"lower right\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.960850Z","iopub.execute_input":"2022-07-17T15:10:23.961295Z","iopub.status.idle":"2022-07-17T15:10:23.978829Z","shell.execute_reply.started":"2022-07-17T15:10:23.961265Z","shell.execute_reply":"2022-07-17T15:10:23.977970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyPipeline(Pipeline):\n    \"\"\"\n    This class calculates categorical and\n    numerical columns in the data and constructs\n    ready to fit sklearn Pipeline object.\n    \"\"\"\n    def __init__(\n        self,\n        data,\n        model,\n        num_imputer=SimpleImputer(strategy=\"mean\"),\n        scaler=StandardScaler(),\n        cat_imputer=SimpleImputer(strategy=\"most_frequent\"),\n        encoder=OneHotEncoder(handle_unknown=\"error\")\n    ):\n        self.data = data\n        self.model = model\n        self.num_imputer = num_imputer\n        self.scaler = scaler\n        self.cat_imputer = cat_imputer\n        self.encoder = encoder\n        \n        numerical_transformer = Pipeline(steps=[\n            (\"imputer\", self.num_imputer),\n            (\"scaler\", self.scaler)\n        ])\n        \n        categorical_transformer = Pipeline(steps=[\n            (\"imputer\", self.cat_imputer),\n            (\"encoder\", self.encoder)\n        ])\n        \n        preprocessor = ColumnTransformer(transformers=[\n            (\"num\", numerical_transformer, self._get_num_cols(self.data)),\n            (\"cat\", categorical_transformer, self._get_cat_cols(self.data))\n        ])\n        \n        super().__init__(steps=[\n            (\"my_preprocessor\", preprocessor),\n            (\"my_model\", self.model)\n        ])\n    \n    @staticmethod\n    def _get_num_cols(X):\n        num_cols = [col for col in X.columns if X[col].dtype in [\"int64\", \"float64\"]]\n        return num_cols\n    \n    @staticmethod\n    def _get_cat_cols(X):\n        cat_cols = [col for col in X.columns if X[col].dtype == \"object\"]\n        return cat_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.980052Z","iopub.execute_input":"2022-07-17T15:10:23.980495Z","iopub.status.idle":"2022-07-17T15:10:23.994365Z","shell.execute_reply.started":"2022-07-17T15:10:23.980467Z","shell.execute_reply":"2022-07-17T15:10:23.993591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_pipeline = MyPipeline(data=X_train, model=XGBClassifier())\nxgb_pipeline","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:23.995550Z","iopub.execute_input":"2022-07-17T15:10:23.996170Z","iopub.status.idle":"2022-07-17T15:10:24.055536Z","shell.execute_reply.started":"2022-07-17T15:10:23.996136Z","shell.execute_reply":"2022-07-17T15:10:24.054860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tune hyperparameters\nparam_grid = {\n    \"model__max_depth\" : [2, 3, 4, 5],\n    \"model__n_estimators\" : [100, 200, 300, 400, 500, 600],\n    \"model__learning_rate\" : [0.2, 0.15, 0.1]\n}\n\nsearch = GridSearchCV(xgb_pipeline, param_grid, scoring='roc_auc', n_jobs=4)\nsearch.fit(X_train, y_train)\nprint(f\"Best XGBClassifier parameters (CV score={search.best_score_:.3f})\")\nprint(search.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:10:24.056399Z","iopub.execute_input":"2022-07-17T15:10:24.057154Z","iopub.status.idle":"2022-07-17T15:12:43.867765Z","shell.execute_reply.started":"2022-07-17T15:10:24.057128Z","shell.execute_reply":"2022-07-17T15:12:43.866951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_parameters = search.best_estimator_.get_params()\nxgb_pipeline.set_params(**best_parameters)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:12:43.870971Z","iopub.execute_input":"2022-07-17T15:12:43.871967Z","iopub.status.idle":"2022-07-17T15:12:43.949165Z","shell.execute_reply.started":"2022-07-17T15:12:43.871909Z","shell.execute_reply":"2022-07-17T15:12:43.947631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 1, figsize=(10, 15))\n\ntitle = \"Learning curves (XGBClassifier)\"\naxes = plot_learning_curve(xgb_pipeline, title, X_train, y_train, axes=axes, ylim=[0.5, 1.03], cv=10, n_jobs=4)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:12:43.950743Z","iopub.execute_input":"2022-07-17T15:12:43.951030Z","iopub.status.idle":"2022-07-17T15:13:05.491303Z","shell.execute_reply.started":"2022-07-17T15:12:43.951003Z","shell.execute_reply":"2022-07-17T15:13:05.490267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_pipeline.fit(X_train, y_train)\ny_pred = xgb_pipeline.predict(X_valid)\n\n# Check classification metrics on validation set\nprint(f\"Accuracy score: {accuracy_score(y_valid, y_pred):.3f}\")\nprint(f\"ROC AUC score: {roc_auc_score(y_valid, y_pred):.3f}\")\ncm = confusion_matrix(y_valid, y_pred, labels=xgb_pipeline.classes_)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=xgb_pipeline.classes_)\ndisp.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:13:05.493106Z","iopub.execute_input":"2022-07-17T15:13:05.493868Z","iopub.status.idle":"2022-07-17T15:13:06.352650Z","shell.execute_reply.started":"2022-07-17T15:13:05.493824Z","shell.execute_reply":"2022-07-17T15:13:06.351162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = xgb_pipeline.predict(X_valid)\nroc_plot(y_valid, pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:13:06.354200Z","iopub.execute_input":"2022-07-17T15:13:06.354658Z","iopub.status.idle":"2022-07-17T15:13:06.541168Z","shell.execute_reply.started":"2022-07-17T15:13:06.354624Z","shell.execute_reply":"2022-07-17T15:13:06.540409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_pipeline = MyPipeline(data=X_train, model=RandomForestClassifier())\nrf_pipeline","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:13:06.542249Z","iopub.execute_input":"2022-07-17T15:13:06.543691Z","iopub.status.idle":"2022-07-17T15:13:06.581398Z","shell.execute_reply.started":"2022-07-17T15:13:06.543619Z","shell.execute_reply":"2022-07-17T15:13:06.580160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid = {\n    \"model__n_estimators\" : [400, 500, 600, 700, 800],\n    \"model__max_depth\" : [2, 3, 5, 10, 20],\n    \"model__min_samples_leaf\" : [3, 4, 5],\n    \"model__min_samples_split\" : [8, 10, 12]\n}\n\nsearch = GridSearchCV(rf_pipeline, param_grid, scoring='roc_auc', n_jobs=4)\nsearch.fit(X_train, y_train)\nprint(f\"Best RandomForestClassifier parameters (CV score={search.best_score_:.3f})\")\nprint(search.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:13:06.583338Z","iopub.execute_input":"2022-07-17T15:13:06.583783Z","iopub.status.idle":"2022-07-17T15:19:36.621080Z","shell.execute_reply.started":"2022-07-17T15:13:06.583749Z","shell.execute_reply":"2022-07-17T15:19:36.619731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_parameters = search.best_estimator_.get_params()\nrf_pipeline.set_params(**best_parameters)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:19:36.622513Z","iopub.execute_input":"2022-07-17T15:19:36.622945Z","iopub.status.idle":"2022-07-17T15:19:36.657710Z","shell.execute_reply.started":"2022-07-17T15:19:36.622908Z","shell.execute_reply":"2022-07-17T15:19:36.656734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 1, figsize=(10, 15))\n\ntitle = \"Learning curves (RandomForestClassifier)\"\naxes = plot_learning_curve(rf_pipeline, title, X_train, y_train, axes=axes, ylim=[0.5, 1.03], cv=10, n_jobs=4)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:19:36.659023Z","iopub.execute_input":"2022-07-17T15:19:36.659710Z","iopub.status.idle":"2022-07-17T15:20:19.981267Z","shell.execute_reply.started":"2022-07-17T15:19:36.659682Z","shell.execute_reply":"2022-07-17T15:20:19.980155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_pipeline.fit(X_train, y_train)\ny_pred = rf_pipeline.predict(X_valid)\n\n# Check classification metrics on validation set\nprint(f\"Accuracy score: {accuracy_score(y_valid, y_pred):.3f}\")\nprint(f\"ROC AUC score: {roc_auc_score(y_valid, y_pred):.3f}\")\ncm = confusion_matrix(y_valid, y_pred, labels=rf_pipeline.classes_)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=rf_pipeline.classes_)\ndisp.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:19.982892Z","iopub.execute_input":"2022-07-17T15:20:19.983313Z","iopub.status.idle":"2022-07-17T15:20:21.222383Z","shell.execute_reply.started":"2022-07-17T15:20:19.983283Z","shell.execute_reply":"2022-07-17T15:20:21.220959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = rf_pipeline.predict(X_valid)\nroc_plot(y_valid, pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:21.223819Z","iopub.execute_input":"2022-07-17T15:20:21.224088Z","iopub.status.idle":"2022-07-17T15:20:21.464945Z","shell.execute_reply.started":"2022-07-17T15:20:21.224063Z","shell.execute_reply":"2022-07-17T15:20:21.463347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc_pipeline = MyPipeline(data=X_train, model=SVC(kernel=\"rbf\"))\nsvc_pipeline","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:21.466757Z","iopub.execute_input":"2022-07-17T15:20:21.469298Z","iopub.status.idle":"2022-07-17T15:20:21.501755Z","shell.execute_reply.started":"2022-07-17T15:20:21.469263Z","shell.execute_reply":"2022-07-17T15:20:21.500676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 1, figsize=(10, 15))\n\ntitle = \"Learning curves (SVC)\"\naxes = plot_learning_curve(svc_pipeline, title, X_train, y_train, axes=axes, ylim=[0.5, 1.03], cv=10, n_jobs=4)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:21.507808Z","iopub.execute_input":"2022-07-17T15:20:21.508167Z","iopub.status.idle":"2022-07-17T15:20:23.701010Z","shell.execute_reply.started":"2022-07-17T15:20:21.508135Z","shell.execute_reply":"2022-07-17T15:20:23.700163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc_pipeline.fit(X_train, y_train)\ny_pred = svc_pipeline.predict(X_valid)\n\n# Check classification metrics on validation set\nprint(f\"Accuracy score: {accuracy_score(y_valid, y_pred):.3f}\")\nprint(f\"ROC AUC score: {roc_auc_score(y_valid, y_pred):.3f}\")\ncm = confusion_matrix(y_valid, y_pred, labels=svc_pipeline.classes_)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=svc_pipeline.classes_)\ndisp.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:23.702737Z","iopub.execute_input":"2022-07-17T15:20:23.703111Z","iopub.status.idle":"2022-07-17T15:20:23.886097Z","shell.execute_reply.started":"2022-07-17T15:20:23.703076Z","shell.execute_reply":"2022-07-17T15:20:23.884413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = svc_pipeline.predict(X_valid)\nroc_plot(y_valid, pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:23.888213Z","iopub.execute_input":"2022-07-17T15:20:23.888929Z","iopub.status.idle":"2022-07-17T15:20:24.059722Z","shell.execute_reply.started":"2022-07-17T15:20:23.888885Z","shell.execute_reply":"2022-07-17T15:20:24.058631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_pipeline = MyPipeline(data=X_train, model=LogisticRegression(solver=\"lbfgs\"))\nlr_pipeline","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:24.061262Z","iopub.execute_input":"2022-07-17T15:20:24.061678Z","iopub.status.idle":"2022-07-17T15:20:24.095250Z","shell.execute_reply.started":"2022-07-17T15:20:24.061644Z","shell.execute_reply":"2022-07-17T15:20:24.093953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 1, figsize=(10, 15))\n\ntitle = \"Learning curves (LinearRegression)\"\naxes = plot_learning_curve(lr_pipeline, title, X_train, y_train, axes=axes, ylim=[0.5, 1.03], cv=10, n_jobs=4)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:24.096767Z","iopub.execute_input":"2022-07-17T15:20:24.097124Z","iopub.status.idle":"2022-07-17T15:20:26.019155Z","shell.execute_reply.started":"2022-07-17T15:20:24.097090Z","shell.execute_reply":"2022-07-17T15:20:26.017854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_pipeline.fit(X_train, y_train)\ny_pred = lr_pipeline.predict(X_valid)\n\n# Check classification metrics on validation set\nprint(f\"Accuracy score: {accuracy_score(y_valid, y_pred):.3f}\")\nprint(f\"ROC AUC score: {roc_auc_score(y_valid, y_pred):.3f}\")\ncm = confusion_matrix(y_valid, y_pred, labels=lr_pipeline.classes_)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=lr_pipeline.classes_)\ndisp.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:26.021257Z","iopub.execute_input":"2022-07-17T15:20:26.021706Z","iopub.status.idle":"2022-07-17T15:20:26.204084Z","shell.execute_reply.started":"2022-07-17T15:20:26.021670Z","shell.execute_reply":"2022-07-17T15:20:26.203292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = lr_pipeline.predict(X_valid)\nroc_plot(y_valid, pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:26.205251Z","iopub.execute_input":"2022-07-17T15:20:26.206522Z","iopub.status.idle":"2022-07-17T15:20:26.367190Z","shell.execute_reply.started":"2022-07-17T15:20:26.206474Z","shell.execute_reply":"2022-07-17T15:20:26.365613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis of XGBClassifier","metadata":{}},{"cell_type":"code","source":"xgb_pipeline = MyPipeline(data=X_train, model=XGBClassifier(), encoder=OrdinalEncoder())\nxgb_pipeline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:26.369206Z","iopub.execute_input":"2022-07-17T15:20:26.369582Z","iopub.status.idle":"2022-07-17T15:20:26.805815Z","shell.execute_reply.started":"2022-07-17T15:20:26.369539Z","shell.execute_reply":"2022-07-17T15:20:26.804501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_importances = pd.Series(xgb_pipeline[\"my_model\"].feature_importances_, index=X_train.columns).sort_values(ascending=True)\n\nfig, ax = plt.subplots(figsize=(8, 6))\nxgb_importances.plot.barh(ax=ax)\nax.set_title(\"Feature importances using MDI\")\nax.set_ylabel(\"Mean decrease in impurity\")\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:26.807101Z","iopub.execute_input":"2022-07-17T15:20:26.807432Z","iopub.status.idle":"2022-07-17T15:20:27.041644Z","shell.execute_reply.started":"2022-07-17T15:20:26.807403Z","shell.execute_reply":"2022-07-17T15:20:27.040551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_new = X_train.drop([\"Parch\", \"SibSp\"], axis=1)\nX_valid_new = X_valid.drop([\"Parch\", \"SibSp\"], axis=1)\n\nxgb_pipeline_new = MyPipeline(data=X_train_new, model=XGBClassifier())","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:27.043747Z","iopub.execute_input":"2022-07-17T15:20:27.044195Z","iopub.status.idle":"2022-07-17T15:20:27.055405Z","shell.execute_reply.started":"2022-07-17T15:20:27.044155Z","shell.execute_reply":"2022-07-17T15:20:27.053707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tune hyperparameters\nparam_grid = {\n    \"model__max_depth\" : [2, 3, 4, 5],\n    \"model__n_estimators\" : [100, 200, 300, 400, 500, 600],\n    \"model__learning_rate\" : [0.2, 0.15, 0.1]\n}\n\nsearch = GridSearchCV(xgb_pipeline_new, param_grid, scoring='roc_auc', n_jobs=4)\nsearch.fit(X_train, y_train)\nprint(f\"Best XGBClassifier parameters (CV score={search.best_score_:.3f})\")\nprint(search.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:20:27.057203Z","iopub.execute_input":"2022-07-17T15:20:27.057591Z","iopub.status.idle":"2022-07-17T15:22:40.259129Z","shell.execute_reply.started":"2022-07-17T15:20:27.057527Z","shell.execute_reply":"2022-07-17T15:22:40.258432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params = search.best_estimator_.get_params()\nxgb_pipeline_new.set_params(**best_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:22:40.262419Z","iopub.execute_input":"2022-07-17T15:22:40.263221Z","iopub.status.idle":"2022-07-17T15:22:40.320135Z","shell.execute_reply.started":"2022-07-17T15:22:40.263180Z","shell.execute_reply":"2022-07-17T15:22:40.319070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 1, figsize=(10, 15))\n\ntitle = \"Learning curves (XGBClassifier)\"\naxes = plot_learning_curve(xgb_pipeline_new, title, X_train, y_train, axes=axes, ylim=[0.5, 1.03], cv=10, n_jobs=4)\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:22:40.321497Z","iopub.execute_input":"2022-07-17T15:22:40.321944Z","iopub.status.idle":"2022-07-17T15:23:11.566594Z","shell.execute_reply.started":"2022-07-17T15:22:40.321916Z","shell.execute_reply":"2022-07-17T15:23:11.565764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_pipeline_new.fit(X_train, y_train)\ny_pred = xgb_pipeline_new.predict(X_valid)\n\n# Check classification metrics on validation set\nprint(f\"Accuracy score: {accuracy_score(y_valid, y_pred):.3f}\")\nprint(f\"ROC AUC score: {roc_auc_score(y_valid, y_pred):.3f}\")\ncm = confusion_matrix(y_valid, y_pred, labels=xgb_pipeline_new.classes_)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=xgb_pipeline_new.classes_)\ndisp.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:11.567915Z","iopub.execute_input":"2022-07-17T15:23:11.568376Z","iopub.status.idle":"2022-07-17T15:23:12.583770Z","shell.execute_reply.started":"2022-07-17T15:23:11.568347Z","shell.execute_reply":"2022-07-17T15:23:12.582701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = xgb_pipeline_new.predict(X_valid)\nroc_plot(y_valid, pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.585052Z","iopub.execute_input":"2022-07-17T15:23:12.585371Z","iopub.status.idle":"2022-07-17T15:23:12.760985Z","shell.execute_reply.started":"2022-07-17T15:23:12.585343Z","shell.execute_reply":"2022-07-17T15:23:12.759739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test dataset","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv(\"../input/titanic/test.csv\")\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.762359Z","iopub.execute_input":"2022-07-17T15:23:12.762714Z","iopub.status.idle":"2022-07-17T15:23:12.792091Z","shell.execute_reply.started":"2022-07-17T15:23:12.762684Z","shell.execute_reply":"2022-07-17T15:23:12.790655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"PassengerId\", \"Name\", \"Ticket\", \"Cabin\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.793944Z","iopub.execute_input":"2022-07-17T15:23:12.794326Z","iopub.status.idle":"2022-07-17T15:23:12.801318Z","shell.execute_reply.started":"2022-07-17T15:23:12.794293Z","shell.execute_reply":"2022-07-17T15:23:12.800174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/titanic/train.csv\")\nX_train = df_train.drop(columns=[\"PassengerId\", \"Name\", \"Ticket\", \"Cabin\"])\ny_train = X_train.pop(\"Survived\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.803472Z","iopub.execute_input":"2022-07-17T15:23:12.803951Z","iopub.status.idle":"2022-07-17T15:23:12.907197Z","shell.execute_reply.started":"2022-07-17T15:23:12.803911Z","shell.execute_reply":"2022-07-17T15:23:12.905714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_pipeline = MyPipeline(data=X_train_new, model=XGBClassifier())","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.908097Z","iopub.status.idle":"2022-07-17T15:23:12.908477Z","shell.execute_reply.started":"2022-07-17T15:23:12.908279Z","shell.execute_reply":"2022-07-17T15:23:12.908296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tune hyperparameters\nparam_grid = {\n    \"model__max_depth\" : [2, 3, 4, 5, 6],\n    \"model__n_estimators\" : [100, 200, 300, 400, 500, 600],\n    \"model__learning_rate\" : [0.3, 0.2, 0.15, 0.1]\n}\n\nsearch = GridSearchCV(final_pipeline, param_grid, scoring='roc_auc', n_jobs=5)\nsearch.fit(X_train, y_train)\nprint(f\"Best XGBClassifier parameters (CV score={search.best_score_:.3f})\")\nprint(search.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.909842Z","iopub.status.idle":"2022-07-17T15:23:12.910209Z","shell.execute_reply.started":"2022-07-17T15:23:12.910028Z","shell.execute_reply":"2022-07-17T15:23:12.910044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params = search.best_estimator_.get_params()\nfinal_pipeline.set_params(**best_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.911339Z","iopub.status.idle":"2022-07-17T15:23:12.911686Z","shell.execute_reply.started":"2022-07-17T15:23:12.911519Z","shell.execute_reply":"2022-07-17T15:23:12.911535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_pipeline.fit(X_train, y_train)\npredictions = final_pipeline.predict(X_test)\ndf_predicts = pd.DataFrame({\"PassengerId\" : df_test[\"PassengerId\"], \"Survived\" : predictions})\ndf_predicts.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.913156Z","iopub.status.idle":"2022-07-17T15:23:12.913444Z","shell.execute_reply.started":"2022-07-17T15:23:12.913297Z","shell.execute_reply":"2022-07-17T15:23:12.913310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_predicts.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T15:23:12.914583Z","iopub.status.idle":"2022-07-17T15:23:12.914878Z","shell.execute_reply.started":"2022-07-17T15:23:12.914734Z","shell.execute_reply":"2022-07-17T15:23:12.914748Z"},"trusted":true},"execution_count":null,"outputs":[]}]}