{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30683,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#import the required libraries\nimport os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.model_selection import cross_val_score, cross_val_predict\nfrom sklearn.model_selection import learning_curve\nfrom sklearn.base import BaseEstimator, ClassifierMixin\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-26T20:08:44.913954Z","iopub.execute_input":"2024-04-26T20:08:44.914724Z","iopub.status.idle":"2024-04-26T20:08:50.376450Z","shell.execute_reply.started":"2024-04-26T20:08:44.914691Z","shell.execute_reply":"2024-04-26T20:08:50.375519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting the directory paths\nROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:08:50.378100Z","iopub.execute_input":"2024-04-26T20:08:50.378551Z","iopub.status.idle":"2024-04-26T20:08:50.383222Z","shell.execute_reply.started":"2024-04-26T20:08:50.378518Z","shell.execute_reply":"2024-04-26T20:08:50.382359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Definitions analysis\ndf_feature = pd.read_csv(ROOT/'feature_definitions.csv')\ndf_feature","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:08:54.712337Z","iopub.execute_input":"2024-04-26T20:08:54.712753Z","iopub.status.idle":"2024-04-26T20:08:54.738239Z","shell.execute_reply.started":"2024-04-26T20:08:54.712725Z","shell.execute_reply":"2024-04-26T20:08:54.737365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Function Definitions\n\nHere we define our utility functions. Functions that allow us to read the data, set the appropriate data types, a function to handle the dates, and also perform feature engineering. (For the feature engineering, we split the month-week column in the base file, and join the tables.\n","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base\n","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:09:03.113779Z","iopub.execute_input":"2024-04-26T20:09:03.114125Z","iopub.status.idle":"2024-04-26T20:09:03.121058Z","shell.execute_reply.started":"2024-04-26T20:09:03.114098Z","shell.execute_reply":"2024-04-26T20:09:03.119971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_nulls(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n        return df\n    \n    @staticmethod\n    def filter_unique(df):\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:09:04.423325Z","iopub.execute_input":"2024-04-26T20:09:04.424008Z","iopub.status.idle":"2024-04-26T20:09:04.436503Z","shell.execute_reply.started":"2024-04-26T20:09:04.423972Z","shell.execute_reply":"2024-04-26T20:09:04.435547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:09:05.692798Z","iopub.execute_input":"2024-04-26T20:09:05.693653Z","iopub.status.idle":"2024-04-26T20:09:05.700939Z","shell.execute_reply.started":"2024-04-26T20:09:05.693621Z","shell.execute_reply":"2024-04-26T20:09:05.699950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Some Background Info on the Data\n\nThis dataset is structured into three levels of depth:\n\n* **Depth 0:** Attributes aggregated at the case_id level. These include characteristics such as the client's age or gender, with one record per case_id.\n* **Depth 1:** Attributes with multiple records per client/application. For instance, data on previous applications or loans in a credit bureau register. Each case_id may have several records, indexed by num_group1.\n* **Depth 2:** This level provides more detailed information for certain attributes from depth 1. For example, for previous applications, details like installment payments or days past due for each payment are available. Each previous application can have zero to multiple records about installment payments, past due days, etc., indexed by num_group2.\n\nIn summary, a single client may have multiple previous applications, and each of these applications can contain multiple records detailing installment payments, past due days, and so on.","metadata":{}},{"cell_type":"markdown","source":"Given this format of data, we need to have some way to aggregate it. To aggregate the data, we have opted to use the max() function.","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:09:11.041506Z","iopub.execute_input":"2024-04-26T20:09:11.042332Z","iopub.status.idle":"2024-04-26T20:09:11.052966Z","shell.execute_reply.started":"2024-04-26T20:09:11.042295Z","shell.execute_reply":"2024-04-26T20:09:11.051803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading the data\n","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:09:14.709995Z","iopub.execute_input":"2024-04-26T20:09:14.711064Z","iopub.status.idle":"2024-04-26T20:11:21.013390Z","shell.execute_reply.started":"2024-04-26T20:09:14.711028Z","shell.execute_reply":"2024-04-26T20:11:21.012489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:11:21.014837Z","iopub.execute_input":"2024-04-26T20:11:21.015109Z","iopub.status.idle":"2024-04-26T20:11:30.656465Z","shell.execute_reply.started":"2024-04-26T20:11:21.015087Z","shell.execute_reply":"2024-04-26T20:11:30.655524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can we, the data we have to handle is pretty massive. It has 1,526,659 rows and 472 colums.\n","metadata":{}},{"cell_type":"code","source":"case_ids = df[\"case_id\"].unique().shuffle(seed=1)\ncase_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.8, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:02.754616Z","iopub.execute_input":"2024-04-26T20:16:02.755350Z","iopub.status.idle":"2024-04-26T20:16:02.900369Z","shell.execute_reply.started":"2024-04-26T20:16:02.755319Z","shell.execute_reply":"2024-04-26T20:16:02.899231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll now split our dataset into training dataset and testing dataset in the ratio (0.8 : 0.2)\n","metadata":{}},{"cell_type":"code","source":"def caseid_filter(case_ids: pl.DataFrame) -> pl.DataFrame:\n    return (\n        df.filter(pl.col(\"case_id\").is_in(case_ids))\n    )","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:06.211822Z","iopub.execute_input":"2024-04-26T20:16:06.212500Z","iopub.status.idle":"2024-04-26T20:16:06.216938Z","shell.execute_reply.started":"2024-04-26T20:16:06.212464Z","shell.execute_reply":"2024-04-26T20:16:06.215913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = caseid_filter(case_ids_train)\ndf_test = caseid_filter(case_ids_test)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:08.393220Z","iopub.execute_input":"2024-04-26T20:16:08.393598Z","iopub.status.idle":"2024-04-26T20:16:10.384368Z","shell.execute_reply.started":"2024-04-26T20:16:08.393569Z","shell.execute_reply":"2024-04-26T20:16:10.383444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:11.802060Z","iopub.execute_input":"2024-04-26T20:16:11.802984Z","iopub.status.idle":"2024-04-26T20:16:11.807708Z","shell.execute_reply.started":"2024-04-26T20:16:11.802940Z","shell.execute_reply":"2024-04-26T20:16:11.806659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:24.926394Z","iopub.execute_input":"2024-04-26T20:16:24.926800Z","iopub.status.idle":"2024-04-26T20:16:30.541601Z","shell.execute_reply.started":"2024-04-26T20:16:24.926771Z","shell.execute_reply":"2024-04-26T20:16:30.540736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"But looking at the data, we can see the there are columns that are almost completelly filled with null values\n","metadata":{}},{"cell_type":"code","source":"# Test data provided by the competition organizers\ndata_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:34.569336Z","iopub.execute_input":"2024-04-26T20:16:34.570015Z","iopub.status.idle":"2024-04-26T20:16:35.192697Z","shell.execute_reply.started":"2024-04-26T20:16:34.569975Z","shell.execute_reply":"2024-04-26T20:16:35.191903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_sample = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", df_test_sample.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:47.830368Z","iopub.execute_input":"2024-04-26T20:16:47.830804Z","iopub.status.idle":"2024-04-26T20:16:47.880856Z","shell.execute_reply.started":"2024-04-26T20:16:47.830777Z","shell.execute_reply":"2024-04-26T20:16:47.879843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that the data they have provided is insufficient for us, hence that is why we created a split from the provided training data\n","metadata":{}},{"cell_type":"code","source":"df_test_sample.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:51.623332Z","iopub.execute_input":"2024-04-26T20:16:51.624314Z","iopub.status.idle":"2024-04-26T20:16:52.012865Z","shell.execute_reply.started":"2024-04-26T20:16:51.624280Z","shell.execute_reply":"2024-04-26T20:16:52.011972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The test data has a similar problem where some columns have only null values.","metadata":{}},{"cell_type":"markdown","source":"Due to some columns having over 95 percent of the fields as null values, they contribute no significant information, and thus, can be dropped, except if it is the target, WEEK_NUM or case_id column as they are necessary for the final result.","metadata":{}},{"cell_type":"code","source":"number_of_columns_before_filter = len(df_train.columns)\ndf_train = Pipeline.filter_nulls(df_train)\nprint(f\"Number of columns dropped: {number_of_columns_before_filter - len(df_train.columns)}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:16:55.698891Z","iopub.execute_input":"2024-04-26T20:16:55.699253Z","iopub.status.idle":"2024-04-26T20:16:56.456421Z","shell.execute_reply.started":"2024-04-26T20:16:55.699221Z","shell.execute_reply":"2024-04-26T20:16:56.455478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have dropped a 100 columns here from the training dataset","metadata":{}},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:00.696016Z","iopub.execute_input":"2024-04-26T20:17:00.696359Z","iopub.status.idle":"2024-04-26T20:17:05.734131Z","shell.execute_reply.started":"2024-04-26T20:17:00.696332Z","shell.execute_reply":"2024-04-26T20:17:05.733259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_nulls = df_train.null_count().sum_horizontal().item()\n\nlabels = [\"Null Values\", \"Non-Null Values\"]\ndata_count = np.array([total_nulls,(df_train.shape[0] * df_train.shape[1]) - total_nulls])\n\nfig, ax = plt.subplots()\nax.pie(data_count, labels=labels, autopct='%1.1f%%')\nplt.title(\"Null Distribution in the Data\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:07.228430Z","iopub.execute_input":"2024-04-26T20:17:07.229083Z","iopub.status.idle":"2024-04-26T20:17:07.427148Z","shell.execute_reply.started":"2024-04-26T20:17:07.229049Z","shell.execute_reply":"2024-04-26T20:17:07.426063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Even after we have filtered the columns with over 95 percent of null values, 30% of our data is empty.\n","metadata":{}},{"cell_type":"markdown","source":"And given the large size of the data, we still have to preprocess it further, and one way to do so is to drop columns of the string datatype that have too many unique values as they usually tend to not have too much information.","metadata":{}},{"cell_type":"markdown","source":"Here we will analyse the number of unique values in each column\n","metadata":{}},{"cell_type":"code","source":"# Total number of unique values in the dataset\nn_unique = 0\nfor col in df_train.columns:\n    n_unique+=df_train.n_unique(col)\n    \nprint(n_unique)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:11.887817Z","iopub.execute_input":"2024-04-26T20:17:11.888157Z","iopub.status.idle":"2024-04-26T20:17:20.598273Z","shell.execute_reply.started":"2024-04-26T20:17:11.888132Z","shell.execute_reply":"2024-04-26T20:17:20.597302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select only the String type columns from the dataset\ndf_n_unique = df_train.select(pl.col(pl.String))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:23.829862Z","iopub.execute_input":"2024-04-26T20:17:23.830573Z","iopub.status.idle":"2024-04-26T20:17:23.835633Z","shell.execute_reply.started":"2024-04-26T20:17:23.830541Z","shell.execute_reply":"2024-04-26T20:17:23.834628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_n_unique = df_n_unique.select(pl.all().n_unique())\ndf_n_unique","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:24.655231Z","iopub.execute_input":"2024-04-26T20:17:24.655548Z","iopub.status.idle":"2024-04-26T20:17:25.166576Z","shell.execute_reply.started":"2024-04-26T20:17:24.655524Z","shell.execute_reply":"2024-04-26T20:17:25.165661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 12))\nax.scatter(df_n_unique, df_n_unique.columns)\nax.set_xscale('log')\nplt.tight_layout()\nplt.title(\"Number of Unique String Values in Each Column\")\nplt.xlabel(\"Number of unique values\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:27.549898Z","iopub.execute_input":"2024-04-26T20:17:27.550721Z","iopub.status.idle":"2024-04-26T20:17:28.965274Z","shell.execute_reply.started":"2024-04-26T20:17:27.550689Z","shell.execute_reply":"2024-04-26T20:17:28.964369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Analysing the data for unique values in columns of the String data type, we can see a number of columns that have many unique string values. These columns tend to be noisy and do not provide any significant information to the model. Therefore, we can safely drop these columns. We opt to use a value of 200 as the threshold to filter the columns.\n","metadata":{}},{"cell_type":"code","source":"total_columns = len(df_train.columns)\ndf_train = Pipeline.filter_unique(df_train)\nprint(f\"Number of columns dropped: {total_columns - len(df_train.columns)}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:32.086396Z","iopub.execute_input":"2024-04-26T20:17:32.087139Z","iopub.status.idle":"2024-04-26T20:17:33.305225Z","shell.execute_reply.started":"2024-04-26T20:17:32.087109Z","shell.execute_reply":"2024-04-26T20:17:33.304230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After dropping the columns, the shape of our data is - ","metadata":{}},{"cell_type":"code","source":"print(df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:35.180566Z","iopub.execute_input":"2024-04-26T20:17:35.180902Z","iopub.status.idle":"2024-04-26T20:17:35.185867Z","shell.execute_reply.started":"2024-04-26T20:17:35.180877Z","shell.execute_reply":"2024-04-26T20:17:35.184822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After filtering the columns having too many null and unique values, we have 361 columns remaining + 1 target column","metadata":{}},{"cell_type":"code","source":"# Dropping the same columns from our test data and test sample provided by organisers\ndf_test = df_test.select([col for col in df_train.columns])\ndf_test_sample = df_test_sample.select([col for col in df_train.columns if col != \"target\"]) ","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:40.622456Z","iopub.execute_input":"2024-04-26T20:17:40.623249Z","iopub.status.idle":"2024-04-26T20:17:40.647865Z","shell.execute_reply.started":"2024-04-26T20:17:40.623212Z","shell.execute_reply":"2024-04-26T20:17:40.647034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.shape)\nprint(df_test.shape)\nprint(df_test_sample.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:41.786103Z","iopub.execute_input":"2024-04-26T20:17:41.786445Z","iopub.status.idle":"2024-04-26T20:17:41.791634Z","shell.execute_reply.started":"2024-04-26T20:17:41.786420Z","shell.execute_reply":"2024-04-26T20:17:41.790667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll now convert the dataframes to pandas as we were only using polars for its quick data loading and filtering.","metadata":{}},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:46.801517Z","iopub.execute_input":"2024-04-26T20:17:46.801850Z","iopub.status.idle":"2024-04-26T20:17:46.807234Z","shell.execute_reply.started":"2024-04-26T20:17:46.801825Z","shell.execute_reply":"2024-04-26T20:17:46.806293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Freeing up some resources - ","metadata":{}},{"cell_type":"code","source":"del(data_store)\ndel(df)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:17:51.435606Z","iopub.execute_input":"2024-04-26T20:17:51.436301Z","iopub.status.idle":"2024-04-26T20:17:51.950129Z","shell.execute_reply.started":"2024-04-26T20:17:51.436269Z","shell.execute_reply":"2024-04-26T20:17:51.949191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:18:05.568249Z","iopub.execute_input":"2024-04-26T20:18:05.568886Z","iopub.status.idle":"2024-04-26T20:18:26.142727Z","shell.execute_reply.started":"2024-04-26T20:18:05.568855Z","shell.execute_reply":"2024-04-26T20:18:26.141712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us analyze how many numerical features we have in the dataset and how many categorical features","metadata":{}},{"cell_type":"code","source":"numerical_features = df_train.select_dtypes(include=\"number\")\ncategorical_features = df_train.select_dtypes(include=\"category\")","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:18:27.935316Z","iopub.execute_input":"2024-04-26T20:18:27.936171Z","iopub.status.idle":"2024-04-26T20:18:28.894385Z","shell.execute_reply.started":"2024-04-26T20:18:27.936140Z","shell.execute_reply":"2024-04-26T20:18:28.893391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 150)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:18:30.241964Z","iopub.execute_input":"2024-04-26T20:18:30.242785Z","iopub.status.idle":"2024-04-26T20:18:30.247366Z","shell.execute_reply.started":"2024-04-26T20:18:30.242754Z","shell.execute_reply":"2024-04-26T20:18:30.246284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Number of Numerical Features: {(numerical_features.shape[1])}\")\nprint(f\"Number of Categorical Features: {categorical_features.shape[1]}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:18:31.589729Z","iopub.execute_input":"2024-04-26T20:18:31.590062Z","iopub.status.idle":"2024-04-26T20:18:31.595156Z","shell.execute_reply.started":"2024-04-26T20:18:31.590038Z","shell.execute_reply":"2024-04-26T20:18:31.594231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:18:33.517159Z","iopub.execute_input":"2024-04-26T20:18:33.517814Z","iopub.status.idle":"2024-04-26T20:18:48.242169Z","shell.execute_reply.started":"2024-04-26T20:18:33.517782Z","shell.execute_reply":"2024-04-26T20:18:48.241242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print out all the unique values in each categorical column\nprint(\"Lists of Unique Values in Each Column\")\nprint(\"=\"*25)\nfor col in categorical_features.columns:\n    print(\"_\"*25)\n    print(f\"\\n{col} : \\n{categorical_features[col].unique()}\\n\")\n    ","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:18:52.120098Z","iopub.execute_input":"2024-04-26T20:18:52.120822Z","iopub.status.idle":"2024-04-26T20:18:52.594609Z","shell.execute_reply.started":"2024-04-26T20:18:52.120791Z","shell.execute_reply":"2024-04-26T20:18:52.593638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The random string of letters and numbers are masked values to protect the privacy of the customers, but if two masked values are equal, then they refer to the same thing.","metadata":{}},{"cell_type":"code","source":"# Freeing up some resources\ndel(numerical_features)\ndel(categorical_features)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:18:54.348957Z","iopub.execute_input":"2024-04-26T20:18:54.349667Z","iopub.status.idle":"2024-04-26T20:18:54.363260Z","shell.execute_reply.started":"2024-04-26T20:18:54.349637Z","shell.execute_reply":"2024-04-26T20:18:54.362319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if there is duplication in the data\n\nprint(\"Train is duplicated:\\t\", df_train[\"case_id\"].duplicated().any())\nprint(\"Train Week Range:\\t\", (df_train[\"WEEK_NUM\"].min(), df_train[\"WEEK_NUM\"].max()))\n\nprint()\n\nprint(\"Test is duplicated:\\t\", df_test[\"case_id\"].duplicated().any())\nprint(\"Test Week Range:\\t\", (df_test[\"WEEK_NUM\"].min(), df_test[\"WEEK_NUM\"].max()))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:19:08.160168Z","iopub.execute_input":"2024-04-26T20:19:08.160883Z","iopub.status.idle":"2024-04-26T20:19:08.190662Z","shell.execute_reply.started":"2024-04-26T20:19:08.160851Z","shell.execute_reply":"2024-04-26T20:19:08.189762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the default rate (0.0 - 1.0) over the weeks\nsns.lineplot(\n    data=df_train,\n    x=\"WEEK_NUM\",\n    y=\"target\",\n)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:19:15.375113Z","iopub.execute_input":"2024-04-26T20:19:15.375466Z","iopub.status.idle":"2024-04-26T20:19:25.971651Z","shell.execute_reply.started":"2024-04-26T20:19:15.375440Z","shell.execute_reply":"2024-04-26T20:19:25.970726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model\n","metadata":{}},{"cell_type":"markdown","source":"We have decided to use two different types of Gradient Boosting Machines - LightGBM and CatBoostClassifier for the problem. We will use the models in a voting ensemble of the models and perform Stratified K Fold cross validation when training the models.","metadata":{}},{"cell_type":"code","source":"class VotingModel(BaseEstimator, ClassifierMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:22:12.965937Z","iopub.execute_input":"2024-04-26T20:22:12.966740Z","iopub.status.idle":"2024-04-26T20:22:12.972971Z","shell.execute_reply.started":"2024-04-26T20:22:12.966709Z","shell.execute_reply":"2024-04-26T20:22:12.972156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining the parameters for the models and initialising variables\ncv = StratifiedGroupKFold(n_splits=5, shuffle=True)\n\nfitted_models_cat = []\nfitted_models_lgb = []\n\ncv_scores_cat = []\ncv_scores_lgb = []\n\nparams = {\n \n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,  \n    \"learning_rate\": 0.05,\n    \"n_estimators\": 2000,  \n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.1,\n    \"reg_lambda\": 10,\n    \"extra_trees\":True,\n    'num_leaves':64,\n    \"device\": \"gpu\", \n    \"verbose\": -1,\n\n\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:22:17.055973Z","iopub.execute_input":"2024-04-26T20:22:17.056334Z","iopub.status.idle":"2024-04-26T20:22:17.063010Z","shell.execute_reply.started":"2024-04-26T20:22:17.056306Z","shell.execute_reply":"2024-04-26T20:22:17.061896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the LightGBM model","metadata":{}},{"cell_type":"code","source":"X = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\ny = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:22:22.492380Z","iopub.execute_input":"2024-04-26T20:22:22.492790Z","iopub.status.idle":"2024-04-26T20:22:23.439704Z","shell.execute_reply.started":"2024-04-26T20:22:22.492760Z","shell.execute_reply":"2024-04-26T20:22:23.438907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_eval_result = {}\nfor idx_train, idx_valid in cv.split(X, y, groups=weeks):\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n    model = lgb.LGBMClassifier(**params)\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        callbacks=[lgb.log_evaluation(100), lgb.early_stopping(100), lgb.record_evaluation(eval_result=model_eval_result)]\n    )\n    y_pred_valid = model.predict_proba(X_valid)[:,1]\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores_lgb.append(auc_score)\n\n    fitted_models_lgb.append(model)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:22:26.756111Z","iopub.execute_input":"2024-04-26T20:22:26.756492Z","iopub.status.idle":"2024-04-26T20:45:28.226106Z","shell.execute_reply.started":"2024-04-26T20:22:26.756462Z","shell.execute_reply":"2024-04-26T20:45:28.225319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the auc score of the last tree during training\nN_trees = list(range(1, len(model_eval_result[\"valid_0\"][\"auc\"]) + 1))\nplt.figure(figsize=(6.4, 4.8))\nax = plt.axes()\nax.plot(N_trees,\n            model_eval_result[\"valid_0\"][\"auc\"],\n            linestyle=\"solid\",\n            color=\"blue\",\n            alpha=1,\n            linewidth=1,\n            marker=\"None\",\n            markeredgecolor=\"red\",\n            markerfacecolor=\"None\",\n            markersize=3,\n           )\nax.set_xlabel(r\"Number of trees\", fontdict={\"fontsize\": 10})\nax.set_ylabel(rf\"AUC\", fontdict={\"fontsize\": 10})\nax.minorticks_on()\n\n# Define grid\nax.grid(visible=True, which=\"major\", color=\"lightgray\", linestyle=\"solid\",\n        linewidth=0.5)\nax.grid(visible=True, which=\"minor\", color=\"lightgray\", linestyle=\"dotted\",\n        linewidth=0.5)\nplt.title(\"Validation Metric (AUC)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:56:45.967476Z","iopub.execute_input":"2024-04-26T20:56:45.967855Z","iopub.status.idle":"2024-04-26T20:56:46.480650Z","shell.execute_reply.started":"2024-04-26T20:56:45.967825Z","shell.execute_reply":"2024-04-26T20:56:46.479541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will look at a tree from our ensemble of LightGBM models","metadata":{}},{"cell_type":"code","source":"graph = lgb.create_tree_digraph(fitted_models_lgb[4], format=\"svg\", orientation=\"vertical\")\ngraph.attr(size='100') \ngraph.render('graph_gbm', format='svg', cleanup=True)\ngraph","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:30:46.539173Z","iopub.execute_input":"2024-04-26T22:30:46.539617Z","iopub.status.idle":"2024-04-26T22:30:48.310094Z","shell.execute_reply.started":"2024-04-26T22:30:46.539586Z","shell.execute_reply":"2024-04-26T22:30:48.309130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the CatBoost Model","metadata":{}},{"cell_type":"code","source":"df_train[cat_cols] = df_train[cat_cols].astype(str)\ndf_test[cat_cols] = df_test[cat_cols].astype(str)\n\nX = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\ny = df_train[\"target\"]\n","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:58:02.617906Z","iopub.execute_input":"2024-04-26T20:58:02.618284Z","iopub.status.idle":"2024-04-26T20:58:09.680147Z","shell.execute_reply.started":"2024-04-26T20:58:02.618254Z","shell.execute_reply":"2024-04-26T20:58:09.679104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier, Pool\n\nn_est=1000\n\nfor idx_train, idx_valid in cv.split(X, y, groups=weeks):\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n    train_pool = Pool(X_train, y_train,cat_features=cat_cols)\n    val_pool = Pool(X_valid, y_valid,cat_features=cat_cols)\n    clf = CatBoostClassifier(\n    eval_metric='AUC',\n    task_type='GPU',\n    learning_rate=0.03,\n    iterations=n_est)\n    random_seed=3107\n    clf.fit(train_pool, eval_set=val_pool,verbose=200, plot=True)\n    fitted_models_cat.append(clf)\n    y_pred_valid = clf.predict_proba(X_valid)[:,1]\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores_cat.append(auc_score)\n    \nprint(\"CV AUC scores: \", cv_scores_cat)\nprint(\"Maximum CV AUC score: \", max(cv_scores_cat))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T21:21:23.845116Z","iopub.execute_input":"2024-04-26T21:21:23.845602Z","iopub.status.idle":"2024-04-26T21:54:33.841860Z","shell.execute_reply.started":"2024-04-26T21:21:23.845570Z","shell.execute_reply":"2024-04-26T21:54:33.840817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting a tree from one of our CatBoost classifiers","metadata":{}},{"cell_type":"code","source":"graph = fitted_models_cat[3].plot_tree(tree_idx=0)\ngraph.attr(size='100') \ngraph.render('graph_cat', format='svg', cleanup=True)\ngraph","metadata":{"execution":{"iopub.status.busy":"2024-04-27T00:50:13.916111Z","iopub.execute_input":"2024-04-27T00:50:13.916479Z","iopub.status.idle":"2024-04-27T00:50:13.942528Z","shell.execute_reply.started":"2024-04-27T00:50:13.916451Z","shell.execute_reply":"2024-04-27T00:50:13.941280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_model = VotingModel(fitted_models_cat)\ngbm_model = VotingModel(fitted_models_lgb)\ncombined_model = VotingModel(fitted_models_cat + fitted_models_lgb)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:28:21.049048Z","iopub.execute_input":"2024-04-26T22:28:21.049904Z","iopub.status.idle":"2024-04-26T22:28:21.054124Z","shell.execute_reply.started":"2024-04-26T22:28:21.049871Z","shell.execute_reply":"2024-04-26T22:28:21.053185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"WEEK_NUM\", \"target\"])\nX_test = X_test.set_index(\"case_id\")\nX_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:33:43.960932Z","iopub.execute_input":"2024-04-26T22:33:43.961815Z","iopub.status.idle":"2024-04-26T22:33:45.657513Z","shell.execute_reply.started":"2024-04-26T22:33:43.961784Z","shell.execute_reply":"2024-04-26T22:33:45.656701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation\n","metadata":{}},{"cell_type":"markdown","source":"For the evaluation criteria, we will be using the metrics provided for the competition i.e. AUC-ROC and Gini Stability metric.\n","metadata":{}},{"cell_type":"markdown","source":"### $$ \\text{stability} = \\text{mean(gini)} + 88.0 \\cdot \\min(0, a) - 0.5 \\cdot \\text{std(residuals)} $$\n\n","metadata":{}},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\n\n\ndef compute_metrics(y_true, y_pred, data=df_test):\n    base = data.copy()\n    base[\"score\"] = np.array(y_pred)\n    return (roc_auc_score(y_true, y_pred), gini_stability(base))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:39:39.079813Z","iopub.execute_input":"2024-04-26T22:39:39.080635Z","iopub.status.idle":"2024-04-26T22:39:39.088306Z","shell.execute_reply.started":"2024-04-26T22:39:39.080605Z","shell.execute_reply":"2024-04-26T22:39:39.087328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = df_test[[\"case_id\",\"target\"]]\ny_true = y_true.set_index(\"case_id\")\ny_true.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:34:19.581185Z","iopub.execute_input":"2024-04-26T22:34:19.581557Z","iopub.status.idle":"2024-04-26T22:34:19.593765Z","shell.execute_reply.started":"2024-04-26T22:34:19.581529Z","shell.execute_reply":"2024-04-26T22:34:19.592642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluating the CatBoost Classifier","metadata":{}},{"cell_type":"code","source":"# Cat Prediction\ny_pred_cat = pd.Series(cat_model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:51:34.291035Z","iopub.execute_input":"2024-04-26T22:51:34.291437Z","iopub.status.idle":"2024-04-26T22:51:45.287565Z","shell.execute_reply.started":"2024-04-26T22:51:34.291394Z","shell.execute_reply":"2024-04-26T22:51:45.286755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_cat, stability_cat = compute_metrics(y_true, y_pred_cat)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:53:43.898522Z","iopub.execute_input":"2024-04-26T22:53:43.899345Z","iopub.status.idle":"2024-04-26T22:53:44.540476Z","shell.execute_reply.started":"2024-04-26T22:53:43.899314Z","shell.execute_reply":"2024-04-26T22:53:44.539461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"AUC-ROC Score using the CatBoostClassifier Model:  {roc_cat}\")\nprint(f\"Stability Score using the CatBoostClassifierModel: {stability_cat}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:56:00.758576Z","iopub.execute_input":"2024-04-26T22:56:00.759372Z","iopub.status.idle":"2024-04-26T22:56:00.763959Z","shell.execute_reply.started":"2024-04-26T22:56:00.759339Z","shell.execute_reply":"2024-04-26T22:56:00.763101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_conv = (y_pred_cat >= 0.5).astype(dtype=\"int32\")\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\ncm = confusion_matrix(\ny_true,\ny_pred_conv\n)\n\ndisp = ConfusionMatrixDisplay(cm).plot(cmap=\"cividis\")\ndisp.ax_.set_title(\"Confusion matrix (CatBoost)\", pad=20);\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-26T23:08:25.757292Z","iopub.execute_input":"2024-04-26T23:08:25.758219Z","iopub.status.idle":"2024-04-26T23:08:26.235659Z","shell.execute_reply.started":"2024-04-26T23:08:25.758185Z","shell.execute_reply":"2024-04-26T23:08:26.234576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluating the LightGBM classifier","metadata":{}},{"cell_type":"code","source":"df_test[cat_cols] = df_test[cat_cols].astype(\"category\")\nX_test = df_test.drop(columns=[\"WEEK_NUM\", \"target\"])\nX_test = X_test.set_index(\"case_id\")\n\ny_pred = pd.Series(gbm_model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T22:56:13.400345Z","iopub.execute_input":"2024-04-26T22:56:13.400703Z","iopub.status.idle":"2024-04-26T22:59:28.678130Z","shell.execute_reply.started":"2024-04-26T22:56:13.400677Z","shell.execute_reply":"2024-04-26T22:59:28.677304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc, stability = compute_metrics(y_true, y_pred)\nprint(f\"AUC-ROC Score using the LightGBM Model:  {roc}\")\nprint(f\"Stability Score using the LightGBM Model: {stability}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-26T23:07:20.576094Z","iopub.execute_input":"2024-04-26T23:07:20.576952Z","iopub.status.idle":"2024-04-26T23:07:21.186648Z","shell.execute_reply.started":"2024-04-26T23:07:20.576905Z","shell.execute_reply":"2024-04-26T23:07:21.185661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_conv = (y_pred >= 0.5).astype(dtype=\"int32\")\n\ncm = confusion_matrix(\ny_true,\ny_pred_conv\n)\n\ndisp = ConfusionMatrixDisplay(cm).plot(cmap=\"cividis\")\ndisp.ax_.set_title(\"Confusion matrix\", pad=20);","metadata":{"execution":{"iopub.status.busy":"2024-04-26T23:00:21.562543Z","iopub.execute_input":"2024-04-26T23:00:21.563533Z","iopub.status.idle":"2024-04-26T23:00:21.874388Z","shell.execute_reply.started":"2024-04-26T23:00:21.563491Z","shell.execute_reply":"2024-04-26T23:00:21.873534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluating the combined model","metadata":{}},{"cell_type":"code","source":"df_test[cat_cols] = df_test[cat_cols].astype(\"category\")\nX_test = df_test.drop(columns=[\"WEEK_NUM\", \"target\"])\nX_test = X_test.set_index(\"case_id\")\n\ny_pred = pd.Series(combined_model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T23:02:09.648787Z","iopub.execute_input":"2024-04-26T23:02:09.649832Z","iopub.status.idle":"2024-04-26T23:05:36.589706Z","shell.execute_reply.started":"2024-04-26T23:02:09.649798Z","shell.execute_reply":"2024-04-26T23:05:36.588869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc, stability = compute_metrics(y_true, y_pred)\nprint(f\"AUC-ROC Score using the combined Model:  {roc}\")\nprint(f\"Stability Score using the combined Model: {stability}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-26T23:31:28.524657Z","iopub.execute_input":"2024-04-26T23:31:28.525617Z","iopub.status.idle":"2024-04-26T23:31:29.172228Z","shell.execute_reply.started":"2024-04-26T23:31:28.525581Z","shell.execute_reply":"2024-04-26T23:31:29.171227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_conv = (y_pred >= 0.5).astype(dtype=\"int32\")\n\ncm = confusion_matrix(\ny_true,\ny_pred_conv\n)\n\ndisp = ConfusionMatrixDisplay(cm).plot(cmap=\"cividis\")\ndisp.ax_.set_title(\"Confusion matrix (Combined Model)\", pad=20);","metadata":{"execution":{"iopub.status.busy":"2024-04-26T23:06:44.486353Z","iopub.execute_input":"2024-04-26T23:06:44.486752Z","iopub.status.idle":"2024-04-26T23:06:44.783132Z","shell.execute_reply.started":"2024-04-26T23:06:44.486721Z","shell.execute_reply":"2024-04-26T23:06:44.782196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}