{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Libraries\n","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import (\n    train_test_split, \n    cross_val_score, \n    StratifiedGroupKFold,\n    StratifiedKFold\n)\nfrom sklearn.preprocessing import ( \n    OneHotEncoder,\n    OrdinalEncoder\n)\nfrom sklearn.compose import (\n    ColumnTransformer,\n)\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom optuna.visualization import (\n    plot_optimization_history, \n    plot_param_importances, \n    plot_slice\n)\nfrom imblearn.combine import SMOTETomek\nfrom imblearn.under_sampling import TomekLinks\nfrom sklearn.metrics import roc_auc_score\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn import set_config\nfrom typing import List, Tuple\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport lightgbm as lgb\nimport xgboost as xgb\nimport pandas as pd\nimport polars as pl\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport optuna\nimport shap\nimport gc","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:29.855865Z","iopub.execute_input":"2024-04-27T07:11:29.856699Z","iopub.status.idle":"2024-04-27T07:11:50.568756Z","shell.execute_reply.started":"2024-04-27T07:11:29.856657Z","shell.execute_reply":"2024-04-27T07:11:50.567979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration\n","metadata":{}},{"cell_type":"code","source":"# Global configurations for sklearn:\nset_config(transform_output=\"pandas\")\n\n# Global configurations for pandas:\npd.set_option(\"display.max_columns\", None)\npd.set_option(\"display.max_rows\", 50)\npd.set_option(\"display.precision\", 3)\npd.set_option(\"display.max_colwidth\", None)\n\n# Global configurations for polars:\npl.Config.activate_decimals(True).set_tbl_hide_column_data_types(True)\npl.Config(\n    **dict(\n        tbl_formatting=\"ASCII_FULL_CONDENSED\",\n        tbl_hide_column_data_types=False,\n        tbl_hide_dataframe_shape=True,\n        fmt_float=\"mixed\",\n        tbl_cell_alignment=\"CENTER\",\n        tbl_hide_dtype_separator=True,\n        tbl_cols=100,\n        tbl_rows=50,\n        fmt_str_lengths=100,\n    )\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:50.570684Z","iopub.execute_input":"2024-04-27T07:11:50.571229Z","iopub.status.idle":"2024-04-27T07:11:50.585425Z","shell.execute_reply.started":"2024-04-27T07:11:50.571202Z","shell.execute_reply":"2024-04-27T07:11:50.584437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/\"\nTRAIN_DIR = ROOT_DIR + \"train/\"\nTEST_DIR = ROOT_DIR + \"test/\"\n\nSEED = 42","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:50.586503Z","iopub.execute_input":"2024-04-27T07:11:50.586792Z","iopub.status.idle":"2024-04-27T07:11:50.596619Z","shell.execute_reply.started":"2024-04-27T07:11:50.586770Z","shell.execute_reply":"2024-04-27T07:11:50.595868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions\n","metadata":{}},{"cell_type":"code","source":"class DataFrameProcessor:\n    \"\"\"Dataframe processing class.\"\"\"\n\n    @staticmethod\n    def convert_types(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"Converts columns' data types for memory.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be processed.\n        \"\"\"\n        for column in df.columns:\n            if column == \"target\":\n                df = df.with_columns(pl.col(column).cast(pl.Int8))\n            elif column in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(column).cast(pl.Int32))\n            elif column == \"date_decision\":\n                df = df.with_columns(pl.col(column).cast(pl.Date))\n            elif column[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(column).cast(pl.Float64))\n            elif column[-1] == \"M\":\n                df = df.with_columns(pl.col(column).cast(pl.String))\n            elif column[-1] == \"D\":\n                df = df.with_columns(pl.col(column).cast(pl.Date))\n\n        return df\n\n    @staticmethod\n    def date_processor(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"Processes the date columns.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be processed.\n        \"\"\"\n        for column in df.columns:\n            if column[-1] == \"D\":\n                df = df.with_columns(pl.col(column) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(column).dt.total_days())\n                df = df.with_columns(pl.col(column).cast(pl.Float32))\n\n            if column == \"date_decision\":\n                df = df.with_columns(\n                    pl.col(column).dt.year().alias(\"year_decision\"),\n                    pl.col(column).dt.month().alias(\"month_decision\"),\n                    pl.col(column).dt.day().alias(\"day_decision\"),\n                    pl.col(column).dt.week().alias(\"week_decision\"),\n                    pl.col(column).dt.weekday().alias(\"weekday_decision\"),\n                    pl.col(column).dt.quarter().alias(\"quarter_decision\"),\n                )\n\n        df = df.drop([\"date_decision\", \"MONTH\"])\n\n        return df\n\n    @staticmethod\n    def delete_nulls_column(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"Deletes columns with more than 80% of null values.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be processed.\n        \"\"\"\n        for column in df.columns:\n            if column not in [\"case_id\", \"target\"]:\n                null_percentage = df[column].is_null().sum() / df.shape[0]\n\n                if null_percentage > 0.80:\n                    df = df.drop(column)\n\n        return df\n    \n    @staticmethod\n    def replace_nulls(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"Replaces all null values in the DataFrame with the string 'Missing'.\n\n        Argument:\n        df (pl.DataFrame): The DataFrame to be processed.\n        \"\"\"\n        # Replace nulls across the DataFrame with 'Missing'\n        return df.fill_null('Missing')\n\n    @staticmethod\n    def drop_duplicates(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"Drops duplicates from the DataFrame.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be processed.\n        \"\"\"\n        df = df.unique(keep=\"first\")\n\n        return df\n\n    @staticmethod\n    def drop_columns_with_too_many_categories(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"Drops columns with more than 100 categories or just 1.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be processed.\n        \"\"\"\n        for column in df.columns:\n            if (column not in [\"target\", \"case_id\"]) & (df[column].dtype == pl.String):\n                categories_count = df[column].n_unique()\n\n                if (categories_count == 1) | (categories_count > 200):\n                    df = df.drop(column)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:50.597916Z","iopub.execute_input":"2024-04-27T07:11:50.598188Z","iopub.status.idle":"2024-04-27T07:11:50.616896Z","shell.execute_reply.started":"2024-04-27T07:11:50.598167Z","shell.execute_reply":"2024-04-27T07:11:50.616018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    \"\"\"Dataframe aggreagating class.\"\"\"\n\n    @staticmethod\n    def max_agg(df: pl.DataFrame) -> pl.Expr:\n        \"\"\"Aggregates the DataFrame by the maximum value.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be aggregated.\n        \"\"\"\n        columns = [column for column in df.columns if column[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(column).alias(f\"max_{column}\") for column in columns]\n\n        return expr_max\n\n    @staticmethod\n    def min_agg(df: pl.DataFrame) -> pl.Expr:\n        \"\"\"Aggregates the DataFrame by the minimum value.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be aggregated.\n        \"\"\"\n        columns = [column for column in df.columns if column[-1] in (\"P\", \"A\")]\n\n        expr_min = [pl.min(column).alias(f\"min_{column}\") for column in columns]\n\n        return expr_min\n\n    @staticmethod\n    def date_agg(df: pl.DataFrame) -> pl.Expr:\n        \"\"\"Aggregates the DataFrame by the date columns.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be aggregated.\n        \"\"\"\n        columns = [column for column in df.columns if column[-1] == \"D\"]\n\n        expr_date = [pl.max(column).alias(f\"max_{column}\") for column in columns]\n\n        return expr_date\n\n    @staticmethod\n    def string_agg(df: pl.DataFrame) -> pl.Expr:\n        \"\"\"Aggregates the DataFrame by the string columns.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be aggregated.\n        \"\"\"\n        columns = [column for column in df.columns if column[-1] == \"M\"]\n\n        expr_string = [pl.max(column).alias(f\"max_{column}\") for column in columns]\n\n        return expr_string\n\n    @staticmethod\n    def others_agg(df: pl.DataFrame) -> pl.Expr:\n        \"\"\"Aggregates the DataFrame by the other columns.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be aggregated.\n        \"\"\"\n        columns = [column for column in df.columns if column in (\"T\", \"L\")]\n\n        expr_others = [pl.max(column).alias(f\"max_{column}\") for column in columns]\n\n        return expr_others\n\n    @staticmethod\n    def count_agg(df: pl.DataFrame) -> pl.Expr:\n        \"\"\"Aggregates the DataFrame by the count of rows.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be aggregated.\n        \"\"\"\n        columns = [column for column in df.columns if \"num_group\" in column]\n\n        expr_max = [pl.max(column).alias(f\"max_{column}\") for column in columns]\n\n        return expr_max\n\n    @staticmethod\n    def agg_expr(df: pl.DataFrame) -> pl.Expr:\n        \"\"\"Aggregates the DataFrame by all the columns.\n\n        Argument:\n        df (polars dataframe): The DataFrame to be aggregated.\n        \"\"\"\n        expr_all = (\n            Aggregator.max_agg(df)\n            + Aggregator.min_agg(df)\n            + Aggregator.date_agg(df)\n            + Aggregator.string_agg(df)\n            + Aggregator.others_agg(df)\n            + Aggregator.count_agg(df)\n        )\n        return expr_all","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:50.620061Z","iopub.execute_input":"2024-04-27T07:11:50.620384Z","iopub.status.idle":"2024-04-27T07:11:50.636390Z","shell.execute_reply.started":"2024-04-27T07:11:50.620355Z","shell.execute_reply":"2024-04-27T07:11:50.635704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path: str, depth: int = 0) -> pl.DataFrame:\n    \"\"\"Read file from the given path.\n\n    Arguments:\n    path (str): The path to the file.\n    depth (int): The depth of the file.\n    \"\"\"\n    df = pl.read_parquet(path)\n    df = df.pipe(DataFrameProcessor.convert_types)\n\n    if depth in (1, 2):\n        df = df.group_by(\"case_id\").agg(Aggregator.agg_expr(df))\n\n    return df\n\n\ndef read_files(path: str, depth: int = 0) -> pl.DataFrame:\n    \"\"\"Read multiple files from the given path and concatenate them vertically.\n\n    Arguments:\n    path (str): The path to the files.\n    depth (int): The depth of the file.\n    \"\"\"\n    df_list = []\n    for one_path in glob(path):\n        df = read_file(one_path, depth)\n        df_list.append(df)\n\n    df = pl.concat(df_list, how=\"vertical_relaxed\")\n    df = df.pipe(DataFrameProcessor.drop_duplicates)\n\n    return df\n\n\ndef merge_dataframe(df_base: pl.DataFrame, **depths: dict) -> pl.DataFrame:\n    \"\"\"Join multiple dataframes together.\n\n    Arguments:\n    df_base (pl.DataFrame): The base DataFrame.\n    **depths (dict of lists of pl.DataFrame): Named groups of DataFrames to join with df_base.\n    \"\"\"\n    index = 0\n    for _, depth_group in depths.items():\n        for df in depth_group:\n            if df is not None:\n                df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{index}\")\n                index += 1\n\n    # Here you would integrate any post-join operations if necessary\n    df_base = df_base.pipe(DataFrameProcessor.date_processor)\n\n    return df_base\n\n\ndef convert_to_pandas_df(\n    polars_df: pl.DataFrame, category_columns: List[str] = None\n) -> Tuple[pd.DataFrame, List[str]]:\n    pandas_df = polars_df.to_pandas()\n\n    # Apply the same set of column to the test set\n    if category_columns is None:\n        category_columns = list(pandas_df.select_dtypes(include=[\"object\"]).columns)\n\n    # Convert to category for ML\n    pandas_df[category_columns] = pandas_df[category_columns].astype(\"category\")\n\n    return pandas_df, category_columns\n\n\ndef memory_optimization(train_data: pd.DataFrame) -> pd.DataFrame:\n    \"\"\" Reduce memory usage of dataframe by modifying type of each column.\n\n    Argument:\n    train_data (pd.DataFrame): The DataFrame to be optimized.\n    \"\"\"\n    base_memory = train_data.memory_usage().sum() / 1024**2\n    print(f'Memory usage of dataframe is {base_memory:.2f} MB')\n\n    for column in train_data.columns:\n        column_type = train_data[column].dtype\n\n        if column_type != 'category':\n            column_value_min = train_data[column].min()\n            column_value_max = train_data[column].max()\n            if str(column_type)[:3] == 'int':\n                if column_value_min > np.iinfo(np.int8).min and column_value_max < np.iinfo(np.int8).max:\n                    train_data[column] = train_data[column].astype(np.int8)\n                elif column_value_min > np.iinfo(np.int16).min and column_value_max < np.iinfo(np.int16).max:\n                    train_data[column] = train_data[column].astype(np.int16)\n                elif column_value_min > np.iinfo(np.int32).min and column_value_max < np.iinfo(np.int32).max:\n                    train_data[column] = train_data[column].astype(np.int32)\n                elif column_value_min > np.iinfo(np.int64).min and column_value_max < np.iinfo(np.int64).max:\n                    train_data[column] = train_data[column].astype(np.int64)  \n            else:\n                if column_value_min > np.finfo(np.float16).min and column_value_max < np.finfo(np.float16).max:\n                    train_data[column] = train_data[column].astype(np.float16)\n                elif column_value_min > np.finfo(np.float32).min and column_value_max < np.finfo(np.float32).max:\n                    train_data[column] = train_data[column].astype(np.float32)\n                else:\n                    train_data[column] = train_data[column].astype(np.float64)\n        else:\n            train_data[column] = train_data[column].astype('category')\n\n    optimized_memory = train_data.memory_usage().sum() / 1024**2\n    optimized_percentage = 100 * (base_memory - optimized_memory) / base_memory\n\n    print(f'Memory usage after optimization is: {optimized_memory:.2f} MB')\n    print(f'Decreased by {optimized_percentage:.1f}%')\n\n    return train_data","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:50.637612Z","iopub.execute_input":"2024-04-27T07:11:50.637937Z","iopub.status.idle":"2024-04-27T07:11:50.658426Z","shell.execute_reply.started":"2024-04-27T07:11:50.637909Z","shell.execute_reply":"2024-04-27T07:11:50.657752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Data\n","metadata":{}},{"cell_type":"code","source":"%%time\n# Test pandas reading time\ntest_pandas_df = pd.read_parquet(TRAIN_DIR + 'train_base.parquet')\ntest_pandas_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:50.659394Z","iopub.execute_input":"2024-04-27T07:11:50.659662Z","iopub.status.idle":"2024-04-27T07:11:50.973451Z","shell.execute_reply.started":"2024-04-27T07:11:50.659640Z","shell.execute_reply":"2024-04-27T07:11:50.972517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_polars_df = pl.read_parquet(TRAIN_DIR + 'train_base.parquet')\ndisplay(test_polars_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:50.974586Z","iopub.execute_input":"2024-04-27T07:11:50.974874Z","iopub.status.idle":"2024-04-27T07:11:51.287743Z","shell.execute_reply.started":"2024-04-27T07:11:50.974849Z","shell.execute_reply":"2024-04-27T07:11:51.286865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Comment: Polars read data faster than pandas, since there are many files, it might be a good idea for me to mitigate to Polars.\n","metadata":{}},{"cell_type":"code","source":"del test_pandas_df, test_polars_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:51.288762Z","iopub.execute_input":"2024-04-27T07:11:51.289024Z","iopub.status.idle":"2024-04-27T07:11:51.464123Z","shell.execute_reply.started":"2024-04-27T07:11:51.289000Z","shell.execute_reply":"2024-04-27T07:11:51.463296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_dict = {\n    \"df_base\": read_file(TRAIN_DIR + \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR + \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR + \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR + \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR + \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR + \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR + \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR + \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR + \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR + \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR + \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR + \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR + \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR + \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR + \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR + \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR + \"train_person_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:11:51.465647Z","iopub.execute_input":"2024-04-27T07:11:51.465981Z","iopub.status.idle":"2024-04-27T07:14:12.399723Z","shell.execute_reply.started":"2024-04-27T07:11:51.465949Z","shell.execute_reply":"2024-04-27T07:14:12.398781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = merge_dataframe(**train_dict)\ntrain_df = train_df.pipe(\n    DataFrameProcessor.delete_nulls_column).pipe(\n        DataFrameProcessor.drop_columns_with_too_many_categories).pipe(\n            DataFrameProcessor.replace_nulls)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:12.401060Z","iopub.execute_input":"2024-04-27T07:14:12.401422Z","iopub.status.idle":"2024-04-27T07:14:24.579610Z","shell.execute_reply.started":"2024-04-27T07:14:12.401389Z","shell.execute_reply":"2024-04-27T07:14:24.578699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_dict = {\n    \"df_base\": read_file(TEST_DIR + \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR + \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR + \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR + \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR + \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR + \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR + \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR + \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR + \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR + \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR + \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR + \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR + \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR + \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR + \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR + \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR + \"test_person_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:24.580796Z","iopub.execute_input":"2024-04-27T07:14:24.581095Z","iopub.status.idle":"2024-04-27T07:14:24.999457Z","shell.execute_reply.started":"2024-04-27T07:14:24.581069Z","shell.execute_reply":"2024-04-27T07:14:24.998542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_df = merge_dataframe(**test_dict)\ntest_df = test_df.select([column for column in train_df.columns if column != \"target\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:25.000668Z","iopub.execute_input":"2024-04-27T07:14:25.001027Z","iopub.status.idle":"2024-04-27T07:14:25.052050Z","shell.execute_reply.started":"2024-04-27T07:14:25.000993Z","shell.execute_reply":"2024-04-27T07:14:25.051223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df, used_category_columns = convert_to_pandas_df(train_df)\ntest_df, _ = convert_to_pandas_df(test_df, used_category_columns)\n\ndel used_category_columns\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:25.055815Z","iopub.execute_input":"2024-04-27T07:14:25.056063Z","iopub.status.idle":"2024-04-27T07:14:39.875920Z","shell.execute_reply.started":"2024-04-27T07:14:25.056043Z","shell.execute_reply":"2024-04-27T07:14:39.875064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = memory_optimization(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:39.877162Z","iopub.execute_input":"2024-04-27T07:14:39.877435Z","iopub.status.idle":"2024-04-27T07:14:45.960333Z","shell.execute_reply.started":"2024-04-27T07:14:39.877412Z","shell.execute_reply":"2024-04-27T07:14:45.959438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_dict, test_dict\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:45.961610Z","iopub.execute_input":"2024-04-27T07:14:45.961927Z","iopub.status.idle":"2024-04-27T07:14:46.427473Z","shell.execute_reply.started":"2024-04-27T07:14:45.961899Z","shell.execute_reply":"2024-04-27T07:14:46.426575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:46.430002Z","iopub.execute_input":"2024-04-27T07:14:46.430400Z","iopub.status.idle":"2024-04-27T07:14:46.693999Z","shell.execute_reply.started":"2024-04-27T07:14:46.430363Z","shell.execute_reply":"2024-04-27T07:14:46.693065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train set shape: {train_df.shape}')","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:46.695403Z","iopub.execute_input":"2024-04-27T07:14:46.695688Z","iopub.status.idle":"2024-04-27T07:14:46.700214Z","shell.execute_reply.started":"2024-04-27T07:14:46.695664Z","shell.execute_reply":"2024-04-27T07:14:46.699272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:46.701391Z","iopub.execute_input":"2024-04-27T07:14:46.702132Z","iopub.status.idle":"2024-04-27T07:14:46.745120Z","shell.execute_reply.started":"2024-04-27T07:14:46.702102Z","shell.execute_reply":"2024-04-27T07:14:46.744199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Time series","metadata":{}},{"cell_type":"code","source":"sns.lineplot(data=train_df,x=\"WEEK_NUM\",y=\"target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:14:46.746306Z","iopub.execute_input":"2024-04-27T07:14:46.746653Z","iopub.status.idle":"2024-04-27T07:15:00.384823Z","shell.execute_reply.started":"2024-04-27T07:14:46.746625Z","shell.execute_reply":"2024-04-27T07:15:00.383918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(data=train_df,x=\"week_decision\",y=\"target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:00.386036Z","iopub.execute_input":"2024-04-27T07:15:00.386330Z","iopub.status.idle":"2024-04-27T07:15:12.667739Z","shell.execute_reply.started":"2024-04-27T07:15:00.386306Z","shell.execute_reply":"2024-04-27T07:15:12.666781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Target variable","metadata":{}},{"cell_type":"code","source":"sns.countplot(data=train_df, x='target')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:12.668831Z","iopub.execute_input":"2024-04-27T07:15:12.669111Z","iopub.status.idle":"2024-04-27T07:15:12.896289Z","shell.execute_reply.started":"2024-04-27T07:15:12.669087Z","shell.execute_reply":"2024-04-27T07:15:12.895307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-processing","metadata":{}},{"cell_type":"code","source":"features = train_df.drop(['case_id', 'target', \"WEEK_NUM\"], axis=1)\ntarget = train_df['target']\nweeks = train_df['WEEK_NUM']\n\ndel train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:12.897988Z","iopub.execute_input":"2024-04-27T07:15:12.898665Z","iopub.status.idle":"2024-04-27T07:15:14.745364Z","shell.execute_reply.started":"2024-04-27T07:15:12.898629Z","shell.execute_reply":"2024-04-27T07:15:14.744388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    features, target, test_size=0.05, random_state=SEED, stratify=target\n)\n\n# Check the shape of the train and test data\nprint(f\"Train data shape: {X_train.shape}, {y_train.shape}\")\nprint(f\"Test data shape: {X_test.shape}, {y_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:14.746623Z","iopub.execute_input":"2024-04-27T07:15:14.746928Z","iopub.status.idle":"2024-04-27T07:15:18.634221Z","shell.execute_reply.started":"2024-04-27T07:15:14.746903Z","shell.execute_reply":"2024-04-27T07:15:18.633285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"# boolean_columns = train_df.select_dtypes(include='bool').columns.tolist()\ncategorical_columns = X_train.select_dtypes(include='category').columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:18.635660Z","iopub.execute_input":"2024-04-27T07:15:18.636281Z","iopub.status.idle":"2024-04-27T07:15:18.697715Z","shell.execute_reply.started":"2024-04-27T07:15:18.636245Z","shell.execute_reply":"2024-04-27T07:15:18.696945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Not used at the moment\nprocessor = ColumnTransformer(\n    transformers=[\n        ('boolean', OrdinalEncoder(), boolean_columns),\n        ('categorical', OneHotEncoder(sparse_output=False, handle_unknown=\"ignore\"), categorical_columns)\n    ], remainder='passthrough'\n)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:18.698804Z","iopub.execute_input":"2024-04-27T07:15:18.699076Z","iopub.status.idle":"2024-04-27T07:15:18.704740Z","shell.execute_reply.started":"2024-04-27T07:15:18.699052Z","shell.execute_reply":"2024-04-27T07:15:18.703896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metric","metadata":{}},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"class_weights = compute_class_weight('balanced', classes=[0, 1], y=target)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:18.705829Z","iopub.execute_input":"2024-04-27T07:15:18.706117Z","iopub.status.idle":"2024-04-27T07:15:18.944965Z","shell.execute_reply.started":"2024-04-27T07:15:18.706082Z","shell.execute_reply":"2024-04-27T07:15:18.944196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_params = {\n    'objective': 'binary:logistic',\n    'eval_metric': 'auc',\n    'device': 'gpu',    \n    'random_state': SEED,\n    'enable_categorical': True\n}\n\ncat_clf = CatBoostClassifier(\n    task_type='GPU',\n    random_state=SEED,\n    loss_function='Logloss',\n    cat_features=categorical_columns,\n    eval_metric='AUC',\n    early_stopping_rounds=10,\n    verbose=False)\n\nlgbm_params = {\n    'objective': 'binary',           \n    'metric': 'auc',                 \n    'device_type': 'gpu',            \n    'random_state': SEED,            \n    'verbosity': -1,                 \n    'num_threads': -1,\n    'is_unbalanced': True\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:18.946031Z","iopub.execute_input":"2024-04-27T07:15:18.946321Z","iopub.status.idle":"2024-04-27T07:15:18.957157Z","shell.execute_reply.started":"2024-04-27T07:15:18.946297Z","shell.execute_reply":"2024-04-27T07:15:18.956288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Cross-validation split\nkfold = StratifiedKFold(shuffle=False) \n\n# StratifiedKFold(shuffle=False)  \n# StratifiedGroupKFold(shuffle=False)\n\nauc_scores = []\n\nfor train_index, valid_index in tqdm(kfold.split(X_train, y_train), total=kfold.get_n_splits(), desc='K-Fold Progress'):\n    y_train_fold = y_train.iloc[train_index]\n    weights_fold = np.where(y_train_fold == 0, class_weights[0], class_weights[1])\n\n    # Create DMatrix for training and validation sets with appropriate weights\n    dtrain = xgb.DMatrix(data=X_train.iloc[train_index], \n                         label=y_train_fold, \n                         enable_categorical=True, \n                         weight=weights_fold)\n    \n    dvalid = xgb.DMatrix(data=X_train.iloc[valid_index], \n                         label=y_train.iloc[valid_index], \n                         enable_categorical=True)\n\n    # Train the model\n    bst = xgb.train(\n        xgb_params,\n        dtrain,\n        num_boost_round=100,\n        evals=[(dvalid, 'validation')],\n        early_stopping_rounds=10,\n        verbose_eval=False\n    )\n\n    # Predict probabilities for the validation set\n    y_pred = bst.predict(dvalid)\n\n    # Calculate AUC and store the score\n    auc = roc_auc_score(y_train.iloc[valid_index], y_pred)\n    auc_scores.append(auc)\n\n# Output the mean AUC\nprint(f'\\nMean AUC: {np.mean(auc_scores)}\\n')","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:15:18.958099Z","iopub.execute_input":"2024-04-27T07:15:18.958340Z","iopub.status.idle":"2024-04-27T07:16:49.510351Z","shell.execute_reply.started":"2024-04-27T07:15:18.958319Z","shell.execute_reply":"2024-04-27T07:16:49.509411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Cross-validation split\nkfold = StratifiedKFold(shuffle=False) \n\n# StratifiedKFold(shuffle=False)  \n# StratifiedGroupKFold(shuffle=False)\n\nauc_scores = []\nfor train_index, valid_index in tqdm(kfold.split(X_train, y_train), total=kfold.get_n_splits(), desc='K-Fold Progress'):\n    train_data = lgb.Dataset(data=X_train.iloc[train_index], label=y_train.iloc[train_index])\n    valid_data = lgb.Dataset(data=X_train.iloc[valid_index], label=y_train.iloc[valid_index])\n\n    # Train the model\n    gbm = lgb.train(\n        lgbm_params,\n        train_data,\n        num_boost_round=100,\n        valid_sets=[train_data, valid_data],\n        valid_names=['train', 'valid'],\n        callbacks=[lgb.early_stopping(stopping_rounds=10)]\n    )\n    \n    y_pred = gbm.predict(X_train.iloc[valid_index], num_iteration=gbm.best_iteration)\n\n    # Calculate AUC and store the score\n    auc = roc_auc_score(y_train.iloc[valid_index], y_pred)\n    auc_scores.append(auc)\n\n# Output the mean AUC\nprint(f'\\nMean AUC: {np.mean(auc_scores)}\\n')","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:16:49.511675Z","iopub.execute_input":"2024-04-27T07:16:49.511950Z","iopub.status.idle":"2024-04-27T07:21:56.497447Z","shell.execute_reply.started":"2024-04-27T07:16:49.511925Z","shell.execute_reply":"2024-04-27T07:21:56.496515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Cross-validation split\nkfold = StratifiedKFold(shuffle=False) \n\n# StratifiedKFold(shuffle=False)  \n# StratifiedGroupKFold(shuffle=False)\n\nfor column in categorical_columns:\n    X_train[column] = X_train[column].cat.add_categories('missing').fillna('missing')\n\nauc_scores = []\nfor train_index, valid_index in tqdm(kfold.split(X_train, y_train), total=kfold.get_n_splits(), desc='K-Fold Progress'):\n    X_train_split, X_valid_split = X_train.iloc[train_index], X_train.iloc[valid_index]\n    y_train_split, y_valid_split = y_train.iloc[train_index], y_train.iloc[valid_index]\n\n    # Train the model\n    cat_clf.fit(X_train_split, y_train_split, \n                eval_set=(X_valid_split, y_valid_split),\n                use_best_model=True)\n    \n    y_pred = cat_clf.predict_proba(X_valid_split)[:, 1]\n\n    auc = roc_auc_score(y_valid_split, y_pred)\n    auc_scores.append(auc)\n\nprint(f'\\nMean AUC: {np.mean(auc_scores)}\\n')","metadata":{"execution":{"iopub.status.busy":"2024-04-27T07:21:56.498985Z","iopub.execute_input":"2024-04-27T07:21:56.499686Z","iopub.status.idle":"2024-04-27T08:00:50.416935Z","shell.execute_reply.started":"2024-04-27T07:21:56.499651Z","shell.execute_reply":"2024-04-27T08:00:50.415794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"lgbm_clf = LGBMClassifier(\n    objective=\"binary\", \n    metric='auc',\n    random_state=SEED, \n    n_jobs=-1, \n    verbosity=-1,\n    is_unbalanced=True\n)\n\nlgbm_clf.fit(X_train, y_train)\ny_pred_lgbm = lgbm_clf.predict_proba(X_test)[:, 1]\nauc_lgbm = roc_auc_score(y_test, y_pred_lgbm)\nprint(f'LGBM AUC: {auc_lgbm}')","metadata":{"execution":{"iopub.status.busy":"2024-04-27T08:00:50.418391Z","iopub.execute_input":"2024-04-27T08:00:50.419229Z","iopub.status.idle":"2024-04-27T08:02:34.655084Z","shell.execute_reply.started":"2024-04-27T08:00:50.419192Z","shell.execute_reply":"2024-04-27T08:02:34.654061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train, y_train, X_test, y_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T08:02:34.656619Z","iopub.execute_input":"2024-04-27T08:02:34.657378Z","iopub.status.idle":"2024-04-27T08:02:34.845319Z","shell.execute_reply.started":"2024-04-27T08:02:34.657341Z","shell.execute_reply":"2024-04-27T08:02:34.844347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction & Submit","metadata":{}},{"cell_type":"code","source":"# Retrain before prediction\nlgbm_clf.fit(features, target)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T08:02:34.846500Z","iopub.execute_input":"2024-04-27T08:02:34.846815Z","iopub.status.idle":"2024-04-27T08:04:28.429423Z","shell.execute_reply.started":"2024-04-27T08:02:34.846790Z","shell.execute_reply":"2024-04-27T08:04:28.428410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.drop([\"case_id\", \"WEEK_NUM\"], axis=1, inplace=True)\npredictions = lgbm_clf.predict_proba(test_df)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2024-04-27T08:11:54.142074Z","iopub.execute_input":"2024-04-27T08:11:54.143089Z","iopub.status.idle":"2024-04-27T08:11:54.206401Z","shell.execute_reply.started":"2024-04-27T08:11:54.143054Z","shell.execute_reply":"2024-04-27T08:11:54.205604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv')\nsub['score'] = predictions\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-27T08:11:58.191921Z","iopub.execute_input":"2024-04-27T08:11:58.192535Z","iopub.status.idle":"2024-04-27T08:11:58.222439Z","shell.execute_reply.started":"2024-04-27T08:11:58.192506Z","shell.execute_reply":"2024-04-27T08:11:58.221720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del features, target, weeks\ndel lgbm_clf\ndel test_df, predictions\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-27T08:12:00.635845Z","iopub.execute_input":"2024-04-27T08:12:00.636729Z","iopub.status.idle":"2024-04-27T08:12:00.819301Z","shell.execute_reply.started":"2024-04-27T08:12:00.636696Z","shell.execute_reply":"2024-04-27T08:12:00.818276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reference\n\n- [Home Credit Baseline](https://www.kaggle.com/code/greysky/home-credit-baseline)\n","metadata":{}}]}