{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7584174,"sourceType":"datasetVersion","datasetId":4414761}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import umap","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:52:23.387153Z","iopub.execute_input":"2024-05-08T18:52:23.387672Z","iopub.status.idle":"2024-05-08T18:53:00.482342Z","shell.execute_reply.started":"2024-05-08T18:52:23.387615Z","shell.execute_reply":"2024-05-08T18:53:00.481016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport optuna\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import IsolationForest\nfrom sklearn.neighbors import LocalOutlierFactor\nfrom sklearn.impute import SimpleImputer \nfrom sklearn.preprocessing import OneHotEncoder, MinMaxScaler, RobustScaler\nfrom sklearn.compose import make_column_transformer\nfrom sklearn.pipeline import Pipeline\nfrom imblearn.ensemble import BalancedRandomForestClassifier\nfrom sklearn.model_selection import train_test_split, cross_val_score, RepeatedStratifiedKFold\nfrom sklearn.metrics import confusion_matrix, classification_report, roc_auc_score, make_scorer, roc_curve\n#from sklearn.base import BaseEstimator, RegressorMixin\n\nimport joblib\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-08T18:53:00.484798Z","iopub.execute_input":"2024-05-08T18:53:00.485765Z","iopub.status.idle":"2024-05-08T18:53:02.847675Z","shell.execute_reply.started":"2024-05-08T18:53:00.485718Z","shell.execute_reply":"2024-05-08T18:53:02.846543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pre-Fitted Voting Model","metadata":{}},{"cell_type":"code","source":"# class VotingModel(BaseEstimator, RegressorMixin):\n#     def __init__(self, estimators):\n#         super().__init__()\n#         self.estimators = estimators\n        \n#     def fit(self, X, y=None):\n#         return self\n    \n#     def predict(self, X):\n#         y_preds = [estimator.predict(X) for estimator in self.estimators]\n#         return np.mean(y_preds, axis=0)\n    \n#     def predict_proba(self, X):\n#         y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n#         return np.mean(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:02.849046Z","iopub.execute_input":"2024-05-08T18:53:02.849790Z","iopub.status.idle":"2024-05-08T18:53:02.855430Z","shell.execute_reply.started":"2024-05-08T18:53:02.849757Z","shell.execute_reply":"2024-05-08T18:53:02.854139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pipeline","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:02.856875Z","iopub.execute_input":"2024-05-08T18:53:02.857247Z","iopub.status.idle":"2024-05-08T18:53:02.874469Z","shell.execute_reply.started":"2024-05-08T18:53:02.857202Z","shell.execute_reply":"2024-05-08T18:53:02.873068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Automatic Aggregation","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:02.877690Z","iopub.execute_input":"2024-05-08T18:53:02.878057Z","iopub.status.idle":"2024-05-08T18:53:02.892334Z","shell.execute_reply.started":"2024-05-08T18:53:02.878027Z","shell.execute_reply":"2024-05-08T18:53:02.891021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### File I/O","metadata":{}},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:02.893588Z","iopub.execute_input":"2024-05-08T18:53:02.893942Z","iopub.status.idle":"2024-05-08T18:53:02.910209Z","shell.execute_reply.started":"2024-05-08T18:53:02.893909Z","shell.execute_reply":"2024-05-08T18:53:02.908546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:02.911783Z","iopub.execute_input":"2024-05-08T18:53:02.912216Z","iopub.status.idle":"2024-05-08T18:53:02.921955Z","shell.execute_reply.started":"2024-05-08T18:53:02.912180Z","shell.execute_reply":"2024-05-08T18:53:02.920962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:02.923560Z","iopub.execute_input":"2024-05-08T18:53:02.923908Z","iopub.status.idle":"2024-05-08T18:53:02.939814Z","shell.execute_reply.started":"2024-05-08T18:53:02.923876Z","shell.execute_reply":"2024-05-08T18:53:02.938369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configuration","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:02.941336Z","iopub.execute_input":"2024-05-08T18:53:02.941788Z","iopub.status.idle":"2024-05-08T18:53:02.952054Z","shell.execute_reply.started":"2024-05-08T18:53:02.941746Z","shell.execute_reply":"2024-05-08T18:53:02.950581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:02.953811Z","iopub.execute_input":"2024-05-08T18:53:02.954332Z","iopub.status.idle":"2024-05-08T18:53:40.456025Z","shell.execute_reply.started":"2024-05-08T18:53:02.954276Z","shell.execute_reply":"2024-05-08T18:53:40.454826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:40.457504Z","iopub.execute_input":"2024-05-08T18:53:40.457836Z","iopub.status.idle":"2024-05-08T18:53:49.717954Z","shell.execute_reply.started":"2024-05-08T18:53:40.457810Z","shell.execute_reply":"2024-05-08T18:53:49.716435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:49.719666Z","iopub.execute_input":"2024-05-08T18:53:49.720111Z","iopub.status.idle":"2024-05-08T18:53:50.171504Z","shell.execute_reply.started":"2024-05-08T18:53:49.720072Z","shell.execute_reply":"2024-05-08T18:53:50.170394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:50.172755Z","iopub.execute_input":"2024-05-08T18:53:50.173067Z","iopub.status.idle":"2024-05-08T18:53:50.212567Z","shell.execute_reply.started":"2024-05-08T18:53:50.173042Z","shell.execute_reply":"2024-05-08T18:53:50.211111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# My Functions","metadata":{}},{"cell_type":"code","source":"# override Optuna's default logging to ERROR only\noptuna.logging.set_verbosity(optuna.logging.ERROR)\n\n# define a logging callback that will report on only new challenger parameter configurations if a\n# trial has usurped the state of 'best conditions'\ndef champion_callback(study, frozen_trial):\n    \"\"\"\n    Logging callback that will report when a new trial iteration improves upon existing\n    best trial values.\n\n    Note: This callback is not intended for use in distributed computing systems such as Spark\n    or Ray due to the micro-batch iterative implementation for distributing trials to a cluster's\n    workers or agents.\n    The race conditions with file system state management for distributed trials will render\n    inconsistent values with this callback.\n    \"\"\"\n\n    winner = study.user_attrs.get(\"winner\", None)\n\n    if study.best_value and winner != study.best_value:\n        study.set_user_attr(\"winner\", study.best_value)\n        if winner:\n            improvement_percent = (abs(winner - study.best_value) / study.best_value) * 100\n            print(\n                f\"Trial {frozen_trial.number} achieved value: {frozen_trial.value} with \"\n                f\"{improvement_percent: .4f}% improvement\"\n            )\n        else:\n            print(f\"Initial trial {frozen_trial.number} achieved value: {frozen_trial.value}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:50.220062Z","iopub.execute_input":"2024-05-08T18:53:50.220711Z","iopub.status.idle":"2024-05-08T18:53:50.230506Z","shell.execute_reply.started":"2024-05-08T18:53:50.220672Z","shell.execute_reply":"2024-05-08T18:53:50.229219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating a personalized confusion matrix\ndef plot_confusion_matrix(y_test, y_pred):\n    \"\"\"\n    Plot a confusion matrix in a heatmap format for better visualization.\n\n    This function receives real y values and predicted y values and create a plot for the confusion matrix.\n\n    Parameters:\n    - y_test (pandas Series): real values of y (target) variable.\n\n    - y_tpred (pandas Series): predicted values of y (target) variable by the model.\n\n    Returns:\n    - figure: confusion matrix plot figure.\n    \"\"\"\n\n    matrix = confusion_matrix(y_test, y_pred)\n\n    fig, (ax1, ax2) = plt.subplots(1,2, figsize=(8,3))\n    fig.suptitle('Confusion Matrix', y=1.1)\n\n    sns.heatmap(matrix, annot=True, fmt='d', cmap='Blues', ax=ax1)\n    ax1.set_xlabel('Predicted Values')\n    ax1.set_ylabel('Real Values')\n\n    # criando mapa de calor com valores relativos\n    sns.heatmap(matrix / matrix.sum(), annot=True, fmt='.2%', cmap='Blues', ax=ax2)\n    ax2.set_xlabel('Predicted Values')\n    ax2.set_ylabel('Real Values')\n    fig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:50.232112Z","iopub.execute_input":"2024-05-08T18:53:50.232544Z","iopub.status.idle":"2024-05-08T18:53:50.247940Z","shell.execute_reply.started":"2024-05-08T18:53:50.232512Z","shell.execute_reply":"2024-05-08T18:53:50.246530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ROC Curve Plot\ndef plot_roc_curve(model, X_test, y_test):\n    \"\"\"\n    Plot of the ROC curve.\n\n    This function receives the model, X_test and the y_test and returns the ROC curve figure.\n\n    Parameters:\n    - model: a trained model.\n\n    - X_test (pandas DataFrame): pandas test DataFrame.\n\n    - y_test (pandas Series): pandas test Series.\n\n    Returns:\n    - figure: ROC curve figure.\n    \"\"\"\n\n    y_pred_probs = model.decision_function(X_test[['target']])\n    fpr, tpr, thresholds = roc_curve(y_test, y_pred_probs)\n    logit_roc_auc = roc_auc_score(y_test, y_pred_probs)\n\n    fig = plt.figure(figsize=(4,2))\n    plt.plot(fpr, tpr, label=f'(área = {round(logit_roc_auc, 2)})')\n    plt.plot([0, 1], [0, 1],'r--')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('ROC Curve')\n    plt.legend(loc=\"lower right\")","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:50.249754Z","iopub.execute_input":"2024-05-08T18:53:50.250199Z","iopub.status.idle":"2024-05-08T18:53:50.264769Z","shell.execute_reply.started":"2024-05-08T18:53:50.250159Z","shell.execute_reply":"2024-05-08T18:53:50.263456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # ROC Curve Plot\n# def plot_roc_curve(model, X_test, y_test):\n#     \"\"\"\n#     Plot of the ROC curve.\n\n#     This function receives the model, X_test and the y_test and returns the ROC curve figure.\n\n#     Parameters:\n#     - model: a trained model.\n\n#     - X_test (pandas DataFrame): pandas test DataFrame.\n\n#     - y_test (pandas Series): pandas test Series.\n\n#     Returns:\n#     - figure: ROC curve figure.\n#     \"\"\"\n\n#     y_pred_probs = model.predict_proba(X_test)[:,1]\n#     fpr, tpr, thresholds = roc_curve(y_test, y_pred_probs)\n#     logit_roc_auc = roc_auc_score(y_test, y_pred_probs)\n\n#     fig = plt.figure(figsize=(4,2))\n#     plt.plot(fpr, tpr, label=f'(área = {round(logit_roc_auc, 2)})')\n#     plt.plot([0, 1], [0, 1],'r--')\n#     plt.xlabel('False Positive Rate')\n#     plt.ylabel('True Positive Rate')\n#     plt.title('ROC Curve')\n#     plt.legend(loc=\"lower right\")","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:50.266888Z","iopub.execute_input":"2024-05-08T18:53:50.267321Z","iopub.status.idle":"2024-05-08T18:53:50.278602Z","shell.execute_reply.started":"2024-05-08T18:53:50.267287Z","shell.execute_reply":"2024-05-08T18:53:50.277378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base[[\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:50.280435Z","iopub.execute_input":"2024-05-08T18:53:50.280920Z","iopub.status.idle":"2024-05-08T18:53:50.296970Z","shell.execute_reply.started":"2024-05-08T18:53:50.280877Z","shell.execute_reply":"2024-05-08T18:53:50.295661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Elimination","metadata":{}},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:50.298770Z","iopub.execute_input":"2024-05-08T18:53:50.299169Z","iopub.status.idle":"2024-05-08T18:53:52.929130Z","shell.execute_reply.started":"2024-05-08T18:53:50.299108Z","shell.execute_reply":"2024-05-08T18:53:52.927766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pandas Conversion","metadata":{}},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:53:52.930389Z","iopub.execute_input":"2024-05-08T18:53:52.930715Z","iopub.status.idle":"2024-05-08T18:54:09.243104Z","shell.execute_reply.started":"2024-05-08T18:53:52.930688Z","shell.execute_reply":"2024-05-08T18:54:09.241664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Garbage Collection","metadata":{}},{"cell_type":"code","source":"del data_store\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:09.245035Z","iopub.execute_input":"2024-05-08T18:54:09.245548Z","iopub.status.idle":"2024-05-08T18:54:09.639250Z","shell.execute_reply.started":"2024-05-08T18:54:09.245512Z","shell.execute_reply":"2024-05-08T18:54:09.637834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA","metadata":{}},{"cell_type":"code","source":"print(\"Train is duplicated:\\t\", df_train[\"case_id\"].duplicated().any())\nprint(\"Train Week Range:\\t\", (df_train[\"WEEK_NUM\"].min(), df_train[\"WEEK_NUM\"].max()))\n\nprint()\n\nprint(\"Test is duplicated:\\t\", df_test[\"case_id\"].duplicated().any())\nprint(\"Test Week Range:\\t\", (df_test[\"WEEK_NUM\"].min(), df_test[\"WEEK_NUM\"].max()))","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:09.640969Z","iopub.execute_input":"2024-05-08T18:54:09.641347Z","iopub.status.idle":"2024-05-08T18:54:09.685569Z","shell.execute_reply.started":"2024-05-08T18:54:09.641312Z","shell.execute_reply":"2024-05-08T18:54:09.684354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_missing = (df_train.isna().sum() / df_train.shape[0]).reset_index()\ndf_missing.columns = ['var','percent']\ndf_missing.sort_values('percent', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:09.687016Z","iopub.execute_input":"2024-05-08T18:54:09.687416Z","iopub.status.idle":"2024-05-08T18:54:10.286933Z","shell.execute_reply.started":"2024-05-08T18:54:09.687386Z","shell.execute_reply":"2024-05-08T18:54:10.285659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping column with more than 30% of missing data\ncolumns_to_maintain = df_missing.query(\"percent < 0.3\")['var'].tolist()\ncolumns_to_maintain_test = columns_to_maintain.copy()\ncolumns_to_maintain_test.remove('target')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:10.288563Z","iopub.execute_input":"2024-05-08T18:54:10.289026Z","iopub.status.idle":"2024-05-08T18:54:10.300391Z","shell.execute_reply.started":"2024-05-08T18:54:10.288986Z","shell.execute_reply":"2024-05-08T18:54:10.299191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train[columns_to_maintain]\ndf_test = df_test[columns_to_maintain_test]","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:10.301989Z","iopub.execute_input":"2024-05-08T18:54:10.302450Z","iopub.status.idle":"2024-05-08T18:54:10.892855Z","shell.execute_reply.started":"2024-05-08T18:54:10.302408Z","shell.execute_reply":"2024-05-08T18:54:10.891710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_cols = df_train.select_dtypes(exclude=np.number).columns.tolist()\nnumerical_cols = df_train.select_dtypes(include=np.number).columns.tolist()\nnumerical_cols.remove('target')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:10.894687Z","iopub.execute_input":"2024-05-08T18:54:10.895146Z","iopub.status.idle":"2024-05-08T18:54:11.452355Z","shell.execute_reply.started":"2024-05-08T18:54:10.895086Z","shell.execute_reply":"2024-05-08T18:54:11.450914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Univariate Analysis","metadata":{}},{"cell_type":"code","source":"del df_missing\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:11.454069Z","iopub.execute_input":"2024-05-08T18:54:11.454653Z","iopub.status.idle":"2024-05-08T18:54:11.826442Z","shell.execute_reply.started":"2024-05-08T18:54:11.454613Z","shell.execute_reply":"2024-05-08T18:54:11.825276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Numerical columns","metadata":{}},{"cell_type":"code","source":"# for col in numerical_cols:\n#     sns.boxplot(data=df_train, x=col)\n#     plt.title(col)\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:11.828176Z","iopub.execute_input":"2024-05-08T18:54:11.828753Z","iopub.status.idle":"2024-05-08T18:54:11.836506Z","shell.execute_reply.started":"2024-05-08T18:54:11.828665Z","shell.execute_reply":"2024-05-08T18:54:11.835203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols_to_drop = [\n#     'dateofbirth_337D',\n#     'days120_123L',\n#     'days180_256L',\n#     'actualdpdtolerance_344P',\n#     'applicationcnt_361L',\n#     'applications30d_658L',\n#     'applicationscnt_1086L',\n#     'applicationscnt_464L',\n#     'applicationscnt_629L',\n#     'applicationscnt_867L',\n#     'clientscnt12m_3712952L',\n#     'clientscnt3m_3712950L',\n#     'clientscnt6m_3712949L',\n#     'clientscnt_100L',\n#     'clientscnt_1022L',\n#     'clientscnt_1071L',\n#     'clientscnt_1130L',\n#     'clientscnt_157L',\n#     'clientscnt_257L',\n#     'clientscnt_304L',\n#     'clientscnt_360L',\n#     'clientscnt_493L',\n#     'clientscnt_533L',\n#     'clientscnt_887L',\n#     'clientscnt_946L',\n#     'cntincpaycont9m_3716944L',\n#     'commnoinclast6m_3546845L',\n#     'currdebtcredtyperange_828A',\n#     'daysoverduetolerancedd_3976961L',\n#     'deferredmnthsnum_166L',\n#     'downpmt_116A',\n#     'mastercontrelectronic_519L',\n#     'mastercontrexist_109L',\n#     'maxannuity_159A',\n#     'maxdpdfrom6mto36m_3546853P',\n#     'maxdpdlast12m_727P',\n#     'maxdpdlast24m_143P',\n#     'maxdpdlast6m_474P',\n#     'maxdpdlast9m_1059P',\n#     'maxdpdtolerance_374P',\n#     'numactivecredschannel_414L',\n#     'numactiverelcontr_750L',\n#     'numcontrs3months_479L',\n#     'numinstlswithdpd10_728L',\n#     'numnotactivated_1143L',\n#     'numpmtchanneldd_318L',\n#     'numrejects9m_859L',\n#     'posfpd10lastmonth_333P',\n#     'posfpd30lastmonth_3976960P',\n#     'posfstqpd30lastmonth_3976962P',\n#     'sellerplacecnt_915L',\n#     'totalsettled_863A',\n#     'max_actualdpd_943P',\n#     'max_credacc_credlmt_575A',\n#     'max_downpmt_134A',\n#     'max_maxdpdtolerance_577P'\n# ]","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:11.838050Z","iopub.execute_input":"2024-05-08T18:54:11.838522Z","iopub.status.idle":"2024-05-08T18:54:11.851049Z","shell.execute_reply.started":"2024-05-08T18:54:11.838490Z","shell.execute_reply":"2024-05-08T18:54:11.849719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in numerical_cols:\n    median = df_train[col].median()\n    df_train[col] = df_train[col].fillna(median)\n    df_test[col] = df_test[col].fillna(median)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:11.852674Z","iopub.execute_input":"2024-05-08T18:54:11.853150Z","iopub.status.idle":"2024-05-08T18:54:16.466477Z","shell.execute_reply.started":"2024-05-08T18:54:11.853087Z","shell.execute_reply":"2024-05-08T18:54:16.465461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical columns","metadata":{}},{"cell_type":"code","source":"# for col in categorical_cols:\n#     sns.countplot(data=df_train, x=col)\n#     plt.title(col)\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:16.467831Z","iopub.execute_input":"2024-05-08T18:54:16.468177Z","iopub.status.idle":"2024-05-08T18:54:16.473106Z","shell.execute_reply.started":"2024-05-08T18:54:16.468148Z","shell.execute_reply":"2024-05-08T18:54:16.471459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols_to_drop = [\n#     'education_88M',\n#     'maritalst_893M',\n#     'lastapprcommoditycat_1041M',\n#     'lastcancelreason_561M',\n#     'lastrejectcommoditycat_161M',\n#     'lastrejectcommodtypec_5251769M',\n#     'lastrejectreason_759M',\n#     'lastrejectreasonclient_4145040M',\n#     'paytype1st_925L',\n#     'paytype_783L',\n#     'max_cancelreason_3545846M',\n#     'max_rejectreason_755M',\n#     'max_rejectreasonclient_4145042M',\n#     'max_education_927M',\n#     'max_empladdr_district_926M',\n#     'max_empladdr_zipcode_114M',\n#     'max_contaddr_matchlist_1032L',\n#     'max_type_25L'\n# ]","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:16.474667Z","iopub.execute_input":"2024-05-08T18:54:16.475065Z","iopub.status.idle":"2024-05-08T18:54:16.486657Z","shell.execute_reply.started":"2024-05-08T18:54:16.475029Z","shell.execute_reply":"2024-05-08T18:54:16.485342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in categorical_cols:\n    df_train[col] = df_train[col].astype('str')\n    df_test[col] = df_test[col].astype('str')\n    \n    mode = df_train[col].mode()[0]\n    \n    df_train[col] = df_train[col].fillna(value=mode)\n    df_test[col] = df_test[col].fillna(value=mode)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:16.488217Z","iopub.execute_input":"2024-05-08T18:54:16.488545Z","iopub.status.idle":"2024-05-08T18:54:43.068284Z","shell.execute_reply.started":"2024-05-08T18:54:16.488517Z","shell.execute_reply":"2024-05-08T18:54:43.066903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# missing_train = df_train.isna().sum().reset_index()\n# missing_train.columns = ['var','count']\n# missing_train['count'] = round(missing_train['count'] / df_train.shape[0], 2)\n# missing_train = missing_train.query(\"count > 0\")\n# missing_train.sort_values('count', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.070294Z","iopub.execute_input":"2024-05-08T18:54:43.070691Z","iopub.status.idle":"2024-05-08T18:54:43.075760Z","shell.execute_reply.started":"2024-05-08T18:54:43.070659Z","shell.execute_reply":"2024-05-08T18:54:43.074465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"train data shape:\\t\", df_train.shape)\n# print(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.077096Z","iopub.execute_input":"2024-05-08T18:54:43.078267Z","iopub.status.idle":"2024-05-08T18:54:43.089918Z","shell.execute_reply.started":"2024-05-08T18:54:43.078228Z","shell.execute_reply":"2024-05-08T18:54:43.088214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_cols = df_train.columns.tolist()\n# train_cols.remove('target')\n# df_test = df_test[train_cols]","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.091537Z","iopub.execute_input":"2024-05-08T18:54:43.091923Z","iopub.status.idle":"2024-05-08T18:54:43.102662Z","shell.execute_reply.started":"2024-05-08T18:54:43.091890Z","shell.execute_reply":"2024-05-08T18:54:43.101306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.lineplot(\n#     data=df_train,\n#     x=\"WEEK_NUM\",\n#     y=\"target\",\n# )\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.104306Z","iopub.execute_input":"2024-05-08T18:54:43.104724Z","iopub.status.idle":"2024-05-08T18:54:43.115005Z","shell.execute_reply.started":"2024-05-08T18:54:43.104689Z","shell.execute_reply":"2024-05-08T18:54:43.113728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bivariate Analysis","metadata":{}},{"cell_type":"code","source":"# for col in numerical_cols:\n#     #plt.figure(figsize=(6,3))\n#     plt.title(f\"{col} x target\")\n#     sns.boxplot(df_train, x='target', y=col)\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.116670Z","iopub.execute_input":"2024-05-08T18:54:43.117178Z","iopub.status.idle":"2024-05-08T18:54:43.127841Z","shell.execute_reply.started":"2024-05-08T18:54:43.117106Z","shell.execute_reply":"2024-05-08T18:54:43.126623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in numerical_cols:\n#     #plt.figure(figsize=(6,3))\n#     plt.title(f\"{col} x target\")\n#     sns.histplot(df_train, x=col, hue='target')\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.129547Z","iopub.execute_input":"2024-05-08T18:54:43.129942Z","iopub.status.idle":"2024-05-08T18:54:43.140399Z","shell.execute_reply.started":"2024-05-08T18:54:43.129910Z","shell.execute_reply":"2024-05-08T18:54:43.139076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in categorical_cols:\n#     plt.title(f\"{col} x target\")\n#     ax = sns.countplot(data=df_train, x=col, hue='target')\n#     sns.despine(left=True, bottom=True)\n#     for label in ax.containers:\n#         ax.bar_label(label)\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.142433Z","iopub.execute_input":"2024-05-08T18:54:43.142821Z","iopub.status.idle":"2024-05-08T18:54:43.150811Z","shell.execute_reply.started":"2024-05-08T18:54:43.142776Z","shell.execute_reply":"2024-05-08T18:54:43.149553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# matrix = df_train[numerical_cols].corr(method='spearman').round(2)\n# plt.figure(figsize=(15,15))\n# sns.heatmap(matrix, vmin=-1, vmax=1, annot=False);","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.162747Z","iopub.execute_input":"2024-05-08T18:54:43.163186Z","iopub.status.idle":"2024-05-08T18:54:43.168784Z","shell.execute_reply.started":"2024-05-08T18:54:43.163149Z","shell.execute_reply":"2024-05-08T18:54:43.167455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing Data","metadata":{}},{"cell_type":"code","source":"#kf = RepeatedStratifiedKFold(n_splits=5, n_repeats=3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.170426Z","iopub.execute_input":"2024-05-08T18:54:43.170896Z","iopub.status.idle":"2024-05-08T18:54:43.182072Z","shell.execute_reply.started":"2024-05-08T18:54:43.170853Z","shell.execute_reply":"2024-05-08T18:54:43.180831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop('target', axis=1)\ny = df_train['target']","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:43.183776Z","iopub.execute_input":"2024-05-08T18:54:43.184940Z","iopub.status.idle":"2024-05-08T18:54:45.329924Z","shell.execute_reply.started":"2024-05-08T18:54:43.184890Z","shell.execute_reply":"2024-05-08T18:54:45.328701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del df_train\n\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:45.331527Z","iopub.execute_input":"2024-05-08T18:54:45.332717Z","iopub.status.idle":"2024-05-08T18:54:45.337183Z","shell.execute_reply.started":"2024-05-08T18:54:45.332679Z","shell.execute_reply":"2024-05-08T18:54:45.335918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X,y, random_state=42, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:45.338632Z","iopub.execute_input":"2024-05-08T18:54:45.338994Z","iopub.status.idle":"2024-05-08T18:54:57.721751Z","shell.execute_reply.started":"2024-05-08T18:54:45.338964Z","shell.execute_reply":"2024-05-08T18:54:57.720560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X, y\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:57.723305Z","iopub.execute_input":"2024-05-08T18:54:57.723780Z","iopub.status.idle":"2024-05-08T18:54:58.372071Z","shell.execute_reply.started":"2024-05-08T18:54:57.723738Z","shell.execute_reply":"2024-05-08T18:54:58.370775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.set_index('case_id')\nX_test = X_test.set_index('case_id')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:54:58.373565Z","iopub.execute_input":"2024-05-08T18:54:58.373908Z","iopub.status.idle":"2024-05-08T18:55:04.076522Z","shell.execute_reply.started":"2024-05-08T18:54:58.373878Z","shell.execute_reply":"2024-05-08T18:55:04.075267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# numerical_cols.remove('case_id')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.078086Z","iopub.execute_input":"2024-05-08T18:55:04.078627Z","iopub.status.idle":"2024-05-08T18:55:04.083875Z","shell.execute_reply.started":"2024-05-08T18:55:04.078591Z","shell.execute_reply":"2024-05-08T18:55:04.082568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transformer = make_column_transformer(\n#     (OneHotEncoder(sparse=False, handle_unknown='ignore', dtype='int'), categorical_cols),\n#     verbose_feature_names_out=False\n#     )\n    \n# # Transforming\n# X_train_transformed = transformer.fit_transform(X_train)\n# # Transformating back\n# X_train_transformed = pd.DataFrame(X_train_transformed, columns=transformer.get_feature_names_out())\n# # One-hot encoding removed an index. Let's put it back:\n# X_train_transformed.index = X_train.index\n# # Joining tables\n# X_train = pd.concat([X_train, X_train_transformed], axis=1)\n# # Dropping old categorical columns\n# X_train.drop(categorical_cols, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.085353Z","iopub.execute_input":"2024-05-08T18:55:04.085710Z","iopub.status.idle":"2024-05-08T18:55:04.104944Z","shell.execute_reply.started":"2024-05-08T18:55:04.085679Z","shell.execute_reply":"2024-05-08T18:55:04.103710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Transforming\n# X_test_transformed = transformer.transform(X_test)\n# # Transformating back\n# X_test_transformed = pd.DataFrame(X_test_transformed, columns=transformer.get_feature_names_out())\n# # One-hot encoding removed an index. Let's put it back:\n# X_test_transformed.index = X_test.index\n# # Joining tables\n# X_test = pd.concat([X_test, X_test_transformed], axis=1)\n# # Dropping old categorical columns\n# X_test.drop(categorical_cols, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.106430Z","iopub.execute_input":"2024-05-08T18:55:04.106769Z","iopub.status.idle":"2024-05-08T18:55:04.117974Z","shell.execute_reply.started":"2024-05-08T18:55:04.106742Z","shell.execute_reply":"2024-05-08T18:55:04.116712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Transforming\n# df_test_transformed = transformer.transform(df_test)\n# # Transformating back\n# df_test_transformed = pd.DataFrame(df_test_transformed, columns=transformer.get_feature_names_out())\n# # One-hot encoding removed an index. Let's put it back:\n# df_test_transformed.index = df_test.index\n# # Joining tables\n# df_test = pd.concat([df_test, df_test_transformed], axis=1)\n# # Dropping old categorical columns\n# df_test.drop(categorical_cols, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.119323Z","iopub.execute_input":"2024-05-08T18:55:04.119733Z","iopub.status.idle":"2024-05-08T18:55:04.130743Z","shell.execute_reply.started":"2024-05-08T18:55:04.119703Z","shell.execute_reply":"2024-05-08T18:55:04.129570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del X_train_transformed\n# del X_test_transformed\n# del df_test_transformed\n\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.132294Z","iopub.execute_input":"2024-05-08T18:55:04.132738Z","iopub.status.idle":"2024-05-08T18:55:04.146423Z","shell.execute_reply.started":"2024-05-08T18:55:04.132697Z","shell.execute_reply":"2024-05-08T18:55:04.145100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Instantiate the RobustScaler\n# scaler = RobustScaler()\n\n# # Apply the scaler to the DataFrame\n# X_train_scaled = pd.DataFrame(scaler.fit_transform(X_train[numerical_cols]), columns=numerical_cols)\n# X_test_scaled = pd.DataFrame(scaler.transform(X_test[numerical_cols]), columns=numerical_cols)\n# df_test_scaled = pd.DataFrame(scaler.transform(df_test[numerical_cols]), columns=numerical_cols)\n\n# X_train_scaled.index = X_train.index\n# X_test_scaled.index = X_test.index\n# df_test_scaled.index = df_test.index","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.147998Z","iopub.execute_input":"2024-05-08T18:55:04.148982Z","iopub.status.idle":"2024-05-08T18:55:04.156991Z","shell.execute_reply.started":"2024-05-08T18:55:04.148927Z","shell.execute_reply":"2024-05-08T18:55:04.155905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del X_train\n# del X_test\n# del df_test\n\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.158683Z","iopub.execute_input":"2024-05-08T18:55:04.160179Z","iopub.status.idle":"2024-05-08T18:55:04.169061Z","shell.execute_reply.started":"2024-05-08T18:55:04.160111Z","shell.execute_reply":"2024-05-08T18:55:04.167906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# selected_columns = ['WEEK_NUM','month_decision','weekday_decision','dateofbirth_337D'\n# ,'days120_123L','days180_256L','days360_512L','days90_310L'\n# ,'firstquarter_103L','fourthquarter_440L','numberofqueries_373L'\n# ,'secondquarter_766L','thirdquarter_1082L','annuity_780A'\n# ,'credamount_770A','daysoverduetolerancedd_3976961L'\n# ,'disbursedcredamount_1113A','eir_270L','interestrate_311L'\n# ,'lastactivateddate_801D','lastapplicationdate_877D'\n# ,'lastapprcredamount_781A','lastapprdate_640D','maxannuity_159A'\n# ,'maxdebt4_972A','maxdpdfrom6mto36m_3546853P','maxdpdlast12m_727P'\n# ,'maxdpdlast24m_143P','maxdpdlast6m_474P','maxdpdlast9m_1059P'\n# ,'maxdpdtolerance_374P','mobilephncnt_593L','monthsannuity_845L'\n# ,'numincomingpmts_3546848L','numinstlallpaidearly3d_817L'\n# ,'numinstlsallpaid_934L','numinstlswithdpd10_728L'\n# ,'numinstlswithoutdpd_562L','numinstpaidearly3d_3546850L'\n# ,'numinstpaidearly_338L','numinstpaidlate1d_3546852L'\n# ,'numinstregularpaid_973L','numrejects9m_859L','pmtnum_254L','price_1097A'\n# ,'totalsettled_863A','max_annuity_853A','max_credamount_590A'\n# ,'max_mainoccupationinc_437A','max_maxdpdtolerance_577P'\n# ,'max_approvaldate_319D','max_creationdate_885D','max_dateactivated_425D'\n# ,'max_firstnonzeroinstldate_307D','max_pmtnum_8L','max_tenor_203L'\n# ,'max_num_group1','max_mainoccupationinc_384A','max_birth_259D']","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.170829Z","iopub.execute_input":"2024-05-08T18:55:04.171307Z","iopub.status.idle":"2024-05-08T18:55:04.183368Z","shell.execute_reply.started":"2024-05-08T18:55:04.171265Z","shell.execute_reply":"2024-05-08T18:55:04.181928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train_scaled = X_train_scaled[selected_columns]\n# X_test_scaled = X_test_scaled[selected_columns]\n# df_test_scaled = df_test_scaled[selected_columns]","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.184985Z","iopub.execute_input":"2024-05-08T18:55:04.185419Z","iopub.status.idle":"2024-05-08T18:55:04.199341Z","shell.execute_reply.started":"2024-05-08T18:55:04.185383Z","shell.execute_reply":"2024-05-08T18:55:04.198176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"Un_component=10 #Nº of UMAP components\nmapper = umap.UMAP().fit(X_train[numerical_cols[1:]], y=y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T18:55:04.200637Z","iopub.execute_input":"2024-05-08T18:55:04.200962Z","iopub.status.idle":"2024-05-08T19:31:02.816967Z","shell.execute_reply.started":"2024-05-08T18:55:04.200934Z","shell.execute_reply":"2024-05-08T19:31:02.814464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"umap_component= mapper.transform(X_train[numerical_cols[1:]])\numap_df = pd.DataFrame(data=umap_component)\n#umap_df = umap_df.join(pd.DataFrame(y_train, columns=['target']))","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:35:01.696424Z","iopub.execute_input":"2024-05-08T19:35:01.696921Z","iopub.status.idle":"2024-05-08T19:35:04.275041Z","shell.execute_reply.started":"2024-05-08T19:35:01.696881Z","shell.execute_reply":"2024-05-08T19:35:04.273738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.relplot(x=\"component_0\", y=\"component_1\", hue=\"target\", data=umap_df, palette=\"Paired\")","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.270004Z","iopub.status.idle":"2024-05-08T19:31:09.270476Z","shell.execute_reply.started":"2024-05-08T19:31:09.270255Z","shell.execute_reply":"2024-05-08T19:31:09.270275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"umap_component_test= mapper.transform(X_test[numerical_cols[1:]])\numap_df_test = pd.DataFrame(data=umap_component_test)\n#umap_df_test = umap_df_test.join(pd.DataFrame(y_test,columns=['target']))","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:35:47.528167Z","iopub.execute_input":"2024-05-08T19:35:47.529362Z","iopub.status.idle":"2024-05-08T19:44:15.819095Z","shell.execute_reply.started":"2024-05-08T19:35:47.529311Z","shell.execute_reply":"2024-05-08T19:44:15.817741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"umap_component_df_test= mapper.transform(df_test[numerical_cols[1:]])\ndf_test = pd.DataFrame(data=umap_component_df_test)\n#umap_df_test = umap_df_test.join(pd.DataFrame(y_test,columns=['target']))","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:44:15.822036Z","iopub.execute_input":"2024-05-08T19:44:15.822547Z","iopub.status.idle":"2024-05-08T19:44:18.707915Z","shell.execute_reply.started":"2024-05-08T19:44:15.822501Z","shell.execute_reply":"2024-05-08T19:44:18.706763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train, X_test\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:44:18.709722Z","iopub.execute_input":"2024-05-08T19:44:18.710225Z","iopub.status.idle":"2024-05-08T19:44:22.387773Z","shell.execute_reply.started":"2024-05-08T19:44:18.710181Z","shell.execute_reply":"2024-05-08T19:44:22.386311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective_rf(trial):\n    # creating hyperparameter grid space\n    params = {\n            'n_estimators':trial.suggest_int('n_estimators', 10, 700),\n            'max_depth':trial.suggest_int('max_depth', 2, 30),\n            'min_samples_leaf':trial.suggest_int('min_samples_leaf', 1, 10),\n            'class_weight':'balanced',\n            'random_state':42,\n            'sampling_strategy':\"all\",\n            'replacement':True,\n            'bootstrap':False\n    }\n\n    # defining model\n    clf = BalancedRandomForestClassifier(**params)\n    clf.fit(umap_df, y_train)\n    \n    y_train_pred = clf.predict(umap_df)\n    \n    roc_auc = roc_auc_score(y_train, y_train_pred)\n    \n    return roc_auc\n    \n#     y_train_pred = clf.predict_proba(X_train_scaled)\n#     y_train_pred = pd.Series([x[1] for x in y_train_pred], index=y_train.index)\n\n#     a = pd.concat([X_train_scaled, y_train, y_train_pred], axis=1, join='inner')\n#     a.columns = X_train_scaled.columns.tolist() + ['target','score']\n#     a['target'] = a['target'].astype(int)\n    \n#     metric = gini_stability(base=a, w_fallingrate=88.0, w_resstd=-0.5)\n    \n#     return metric","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:54:03.686568Z","iopub.execute_input":"2024-05-08T19:54:03.687112Z","iopub.status.idle":"2024-05-08T19:54:03.696420Z","shell.execute_reply.started":"2024-05-08T19:54:03.687081Z","shell.execute_reply":"2024-05-08T19:54:03.695196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction='maximize')\nstudy.optimize(objective_rf, n_trials=10, callbacks=[champion_callback])\n\ntrial = study.best_trial\nbest_params = trial.params","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:54:05.953413Z","iopub.execute_input":"2024-05-08T19:54:05.954244Z","iopub.status.idle":"2024-05-08T20:55:07.255541Z","shell.execute_reply.started":"2024-05-08T19:54:05.954206Z","shell.execute_reply":"2024-05-08T20:55:07.253825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = BalancedRandomForestClassifier(**best_params)\nmodel.fit(umap_df, y_train)\n\nprint(\"Training Data\")\n\n# making predictions\ny_train_pred = model.predict(umap_df)\n\nprint(classification_report(y_train, y_train_pred, zero_division=0.0))\n\nprint(\"Testing Data\")\n\n# making predictions\ny_test_pred = model.predict(umap_df_test)\n\nprint(classification_report(y_test, y_test_pred, zero_division=0.0))","metadata":{"execution":{"iopub.status.busy":"2024-05-08T20:55:07.258957Z","iopub.execute_input":"2024-05-08T20:55:07.259508Z","iopub.status.idle":"2024-05-08T21:05:42.126094Z","shell.execute_reply.started":"2024-05-08T20:55:07.259453Z","shell.execute_reply":"2024-05-08T21:05:42.124752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_confusion_matrix(y_train, y_train_pred)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T21:05:42.127592Z","iopub.execute_input":"2024-05-08T21:05:42.127959Z","iopub.status.idle":"2024-05-08T21:05:43.018601Z","shell.execute_reply.started":"2024-05-08T21:05:42.127928Z","shell.execute_reply":"2024-05-08T21:05:43.017155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_confusion_matrix(y_test, y_test_pred)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T21:05:43.020991Z","iopub.execute_input":"2024-05-08T21:05:43.021416Z","iopub.status.idle":"2024-05-08T21:05:43.684346Z","shell.execute_reply.started":"2024-05-08T21:05:43.021379Z","shell.execute_reply":"2024-05-08T21:05:43.683072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_roc_curve(lr, umap_df_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.290942Z","iopub.status.idle":"2024-05-08T19:31:09.291364Z","shell.execute_reply.started":"2024-05-08T19:31:09.291158Z","shell.execute_reply":"2024-05-08T19:31:09.291180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Selection","metadata":{}},{"cell_type":"code","source":"# from sklearn.feature_selection import SelectFromModel\n# # from sklearn.model_selection import StratifiedKFold\n\n# selector = SelectFromModel(estimator=model, prefit=True)\n# selector.fit(X_train, y_train)\n# cols_selected = selector.get_feature_names_out()\n# print(cols_selected)\n\n# #X_train_scaled = X_train_scaled[cols_selected]\n# #X_test_scaled = X_test_scaled[cols_selected]\n\n\n# # rfecv = RFECV(estimator=model, step=1, cv=StratifiedKFold(n_splits=2), scoring='roc_auc')\n# # rfecv.fit(X_train_scaled, y_train)\n# # #Selected features\n# # print(X_train_scaled.columns[rfecv.get_support()])\n# # print(\"Optimal number of features : %d\" % rfecv.n_features_)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.292550Z","iopub.status.idle":"2024-05-08T19:31:09.292928Z","shell.execute_reply.started":"2024-05-08T19:31:09.292743Z","shell.execute_reply":"2024-05-08T19:31:09.292758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# study = optuna.create_study(direction='maximize')\n# study.optimize(objective_toll_rf, n_trials=30, callbacks=[champion_callback])\n\n# trial = study.best_trial\n# best_params = trial.params\n\n# model = BalancedRandomForestClassifier(**best_params)\n# model.fit(X_train_scaled, y_train)\n\n# print(\"Training Data\")\n\n# # making predictions\n# y_train_pred = model.predict(X_train_scaled)\n\n# print(classification_report(y_train, y_train_pred, zero_division=0.0))\n\n# print(\"Testing Data\")\n\n# # making predictions\n# y_test_pred = model.predict(X_test_scaled)\n\n# print(classification_report(y_test, y_test_pred, zero_division=0.0))","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.294184Z","iopub.status.idle":"2024-05-08T19:31:09.294592Z","shell.execute_reply.started":"2024-05-08T19:31:09.294394Z","shell.execute_reply":"2024-05-08T19:31:09.294411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_confusion_matrix(y_train, y_train_pred)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.295946Z","iopub.status.idle":"2024-05-08T19:31:09.296366Z","shell.execute_reply.started":"2024-05-08T19:31:09.296159Z","shell.execute_reply":"2024-05-08T19:31:09.296182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_confusion_matrix(y_test, y_test_pred)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.298012Z","iopub.status.idle":"2024-05-08T19:31:09.298583Z","shell.execute_reply.started":"2024-05-08T19:31:09.298291Z","shell.execute_reply":"2024-05-08T19:31:09.298313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_roc_curve(model, X_test_scaled, y_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.301040Z","iopub.status.idle":"2024-05-08T19:31:09.301629Z","shell.execute_reply.started":"2024-05-08T19:31:09.301331Z","shell.execute_reply":"2024-05-08T19:31:09.301353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction","metadata":{}},{"cell_type":"code","source":"# X_test = df_test.drop(columns=[\"WEEK_NUM\"])\n# X_test = X_test.set_index(\"case_id\")\n\n# y_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.303773Z","iopub.status.idle":"2024-05-08T19:31:09.304225Z","shell.execute_reply.started":"2024-05-08T19:31:09.303976Z","shell.execute_reply":"2024-05-08T19:31:09.303991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"#df_test = df_test.set_index('case_id')\ny_test_pred = pd.DataFrame(model.predict_proba(df_test), index=df_test.index)\ny_test_pred = y_test_pred.iloc[:,1]\ny_test_pred","metadata":{"execution":{"iopub.status.busy":"2024-05-08T21:07:54.469361Z","iopub.execute_input":"2024-05-08T21:07:54.469875Z","iopub.status.idle":"2024-05-08T21:07:54.540771Z","shell.execute_reply.started":"2024-05-08T21:07:54.469843Z","shell.execute_reply":"2024-05-08T21:07:54.539519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_test_pred.tolist()\ndf_subm","metadata":{"execution":{"iopub.status.busy":"2024-05-08T21:07:55.896077Z","iopub.execute_input":"2024-05-08T21:07:55.896582Z","iopub.status.idle":"2024-05-08T21:07:55.930713Z","shell.execute_reply.started":"2024-05-08T21:07:55.896547Z","shell.execute_reply":"2024-05-08T21:07:55.929493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-08T19:31:09.310745Z","iopub.status.idle":"2024-05-08T19:31:09.311230Z","shell.execute_reply.started":"2024-05-08T19:31:09.310973Z","shell.execute_reply":"2024-05-08T19:31:09.311000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}