{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-27T07:28:24.251719Z","iopub.execute_input":"2024-05-27T07:28:24.252512Z","iopub.status.idle":"2024-05-27T07:28:25.358393Z","shell.execute_reply.started":"2024-05-27T07:28:24.252471Z","shell.execute_reply":"2024-05-27T07:28:25.357212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom pathlib import Path\nfrom datetime import datetime\nimport gc\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom glob import glob\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport tensorflow as tf\nimport lightgbm as lgb\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:35:40.090618Z","iopub.execute_input":"2024-05-27T07:35:40.091039Z","iopub.status.idle":"2024-05-27T07:35:41.009883Z","shell.execute_reply.started":"2024-05-27T07:35:40.091006Z","shell.execute_reply":"2024-05-27T07:35:41.008725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:28:40.288898Z","iopub.execute_input":"2024-05-27T07:28:40.289272Z","iopub.status.idle":"2024-05-27T07:28:40.301659Z","shell.execute_reply.started":"2024-05-27T07:28:40.289242Z","shell.execute_reply":"2024-05-27T07:28:40.300536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs  ","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:28:40.305234Z","iopub.execute_input":"2024-05-27T07:28:40.305941Z","iopub.status.idle":"2024-05-27T07:28:40.342484Z","shell.execute_reply.started":"2024-05-27T07:28:40.305900Z","shell.execute_reply":"2024-05-27T07:28:40.341422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:28:40.343568Z","iopub.execute_input":"2024-05-27T07:28:40.343896Z","iopub.status.idle":"2024-05-27T07:28:40.357103Z","shell.execute_reply.started":"2024-05-27T07:28:40.343869Z","shell.execute_reply":"2024-05-27T07:28:40.356080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:28:40.358559Z","iopub.execute_input":"2024-05-27T07:28:40.359050Z","iopub.status.idle":"2024-05-27T07:28:40.367622Z","shell.execute_reply.started":"2024-05-27T07:28:40.359013Z","shell.execute_reply":"2024-05-27T07:28:40.366717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:28:40.368932Z","iopub.execute_input":"2024-05-27T07:28:40.369269Z","iopub.status.idle":"2024-05-27T07:28:40.384593Z","shell.execute_reply.started":"2024-05-27T07:28:40.369243Z","shell.execute_reply":"2024-05-27T07:28:40.383582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            #print(col)\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:28:40.385905Z","iopub.execute_input":"2024-05-27T07:28:40.386285Z","iopub.status.idle":"2024-05-27T07:28:40.400916Z","shell.execute_reply.started":"2024-05-27T07:28:40.386249Z","shell.execute_reply":"2024-05-27T07:28:40.399509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:28:40.403393Z","iopub.execute_input":"2024-05-27T07:28:40.404008Z","iopub.status.idle":"2024-05-27T07:28:40.417598Z","shell.execute_reply.started":"2024-05-27T07:28:40.403952Z","shell.execute_reply":"2024-05-27T07:28:40.416400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:32:28.120076Z","iopub.execute_input":"2024-05-27T07:32:28.120477Z","iopub.status.idle":"2024-05-27T07:32:53.343929Z","shell.execute_reply.started":"2024-05-27T07:32:28.120448Z","shell.execute_reply":"2024-05-27T07:32:53.343025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:32:53.346180Z","iopub.execute_input":"2024-05-27T07:32:53.346598Z","iopub.status.idle":"2024-05-27T07:33:00.166949Z","shell.execute_reply.started":"2024-05-27T07:32:53.346562Z","shell.execute_reply":"2024-05-27T07:33:00.165882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        #read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n                \n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:00.168403Z","iopub.execute_input":"2024-05-27T07:33:00.168814Z","iopub.status.idle":"2024-05-27T07:33:00.491317Z","shell.execute_reply.started":"2024-05-27T07:33:00.168779Z","shell.execute_reply":"2024-05-27T07:33:00.490417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:00.493475Z","iopub.execute_input":"2024-05-27T07:33:00.493791Z","iopub.status.idle":"2024-05-27T07:33:00.524828Z","shell.execute_reply.started":"2024-05-27T07:33:00.493765Z","shell.execute_reply":"2024-05-27T07:33:00.523756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:19.565711Z","iopub.execute_input":"2024-05-27T07:33:19.567964Z","iopub.status.idle":"2024-05-27T07:33:22.502293Z","shell.execute_reply.started":"2024-05-27T07:33:19.567901Z","shell.execute_reply":"2024-05-27T07:33:22.501159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:22.504371Z","iopub.execute_input":"2024-05-27T07:33:22.505242Z","iopub.status.idle":"2024-05-27T07:33:37.908987Z","shell.execute_reply.started":"2024-05-27T07:33:22.505210Z","shell.execute_reply":"2024-05-27T07:33:37.907922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:37.910207Z","iopub.execute_input":"2024-05-27T07:33:37.910537Z","iopub.status.idle":"2024-05-27T07:33:38.267003Z","shell.execute_reply.started":"2024-05-27T07:33:37.910509Z","shell.execute_reply":"2024-05-27T07:33:38.265919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_encode_cols = df_train.dtypes[df_train.dtypes == 'category']\none_hot_encode_cols = one_hot_encode_cols.index.to_list()\none_hot_encode_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:38.268504Z","iopub.execute_input":"2024-05-27T07:33:38.268807Z","iopub.status.idle":"2024-05-27T07:33:38.282285Z","shell.execute_reply.started":"2024-05-27T07:33:38.268781Z","shell.execute_reply":"2024-05-27T07:33:38.280947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\n# Initialize the OrdinalEncoder\nencoder = OrdinalEncoder()\n\n# Fit and transform the encoder on the column_to_encode\nfor i in one_hot_encode_cols:\n    df_train[i].fillna(df_train[i].mode()[0], inplace=True)\n    encoded_column = encoder.fit_transform(df_train[[i]])\n\n    # Replace the original column with the encoded values\n    df_train[i] = encoded_column","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:38.285176Z","iopub.execute_input":"2024-05-27T07:33:38.285598Z","iopub.status.idle":"2024-05-27T07:33:56.644842Z","shell.execute_reply.started":"2024-05-27T07:33:38.285561Z","shell.execute_reply":"2024-05-27T07:33:56.643667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\n# Initialize the OrdinalEncoder\nencoder = OrdinalEncoder()\n\n# Fit and transform the encoder on the column_to_encode\nfor i in one_hot_encode_cols:\n    #df_test[i].fillna(df_test[i].mode()[0], inplace=True)\n    encoded_column = encoder.fit_transform(df_test[[i]])\n\n    # Replace the original column with the encoded values\n    df_test[i] = encoded_column","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:56.646590Z","iopub.execute_input":"2024-05-27T07:33:56.647526Z","iopub.status.idle":"2024-05-27T07:33:56.750406Z","shell.execute_reply.started":"2024-05-27T07:33:56.647485Z","shell.execute_reply":"2024-05-27T07:33:56.749064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nc = 0\nprint(c,i)\n\nfor i in df_test:\n    try:\n        if df_test[i].isnull().mean() > 0.95:\n            df_test[i].fillna(0.00, inplace=True)\n        else:\n            df_test[i].fillna(df_test[i].mode()[0], inplace=True)\n    except Exception as e:\n        print(f\"An error occurred while processing column {i}: {e}\")\n \n    if df_test[i].isnull().sum() > 0:\n        print(df_test[i].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:56.752067Z","iopub.execute_input":"2024-05-27T07:33:56.752800Z","iopub.status.idle":"2024-05-27T07:33:56.957660Z","shell.execute_reply.started":"2024-05-27T07:33:56.752757Z","shell.execute_reply":"2024-05-27T07:33:56.956514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nc = 0\nprint(c,i)\n\nfor i in df_train:\n    c+=1\n    try:\n        df_train[i].fillna(df_train[i].mode()[0], inplace=True)\n    except Exception as e:\n        print(f\"An error occurred while processing column {i}: {e}\")\n        \n    if (df_train[i].isnull().sum()) > 0:\n        print(df_train[i].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:33:56.959045Z","iopub.execute_input":"2024-05-27T07:33:56.959467Z","iopub.status.idle":"2024-05-27T07:34:06.001439Z","shell.execute_reply.started":"2024-05-27T07:33:56.959431Z","shell.execute_reply":"2024-05-27T07:34:06.000273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = reduce_mem_usage(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:34:06.002711Z","iopub.execute_input":"2024-05-27T07:34:06.003149Z","iopub.status.idle":"2024-05-27T07:34:10.602388Z","shell.execute_reply.started":"2024-05-27T07:34:06.003109Z","shell.execute_reply":"2024-05-27T07:34:10.601266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = reduce_mem_usage(df_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:34:10.603622Z","iopub.execute_input":"2024-05-27T07:34:10.603938Z","iopub.status.idle":"2024-05-27T07:34:10.737609Z","shell.execute_reply.started":"2024-05-27T07:34:10.603910Z","shell.execute_reply":"2024-05-27T07:34:10.736548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape, df_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:34:10.739119Z","iopub.execute_input":"2024-05-27T07:34:10.739539Z","iopub.status.idle":"2024-05-27T07:34:10.747103Z","shell.execute_reply.started":"2024-05-27T07:34:10.739501Z","shell.execute_reply":"2024-05-27T07:34:10.745905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import resample\n\ndef oversample_minority(X, y):\n    X['y'] = y\n    # Separate majority and minority classes\n    majority = X[X.y == 0]\n    minority = X[X.y == 1]\n    print(minority.shape, majority.shape)\n    \n    # Upsample minority class\n    minority_upsampled = resample(minority,\n                                  replace=True,  # sample with replacement\n                                  n_samples=len(majority),  # to match majority class\n                                  random_state=123)  # reproducible results\n    \n    # Combine majority class with upsampled minority class\n    upsampled = pd.concat([majority, minority_upsampled])\n    \n    y = upsampled.y\n    X = upsampled.drop('y', axis=1)\n    \n    return X, y","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:34:10.748834Z","iopub.execute_input":"2024-05-27T07:34:10.749994Z","iopub.status.idle":"2024-05-27T07:34:10.758482Z","shell.execute_reply.started":"2024-05-27T07:34:10.749936Z","shell.execute_reply":"2024-05-27T07:34:10.757326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train\ny = df_train[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:34:10.759779Z","iopub.execute_input":"2024-05-27T07:34:10.760123Z","iopub.status.idle":"2024-05-27T07:34:10.771877Z","shell.execute_reply.started":"2024-05-27T07:34:10.760094Z","shell.execute_reply":"2024-05-27T07:34:10.770801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_train\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:34:10.775550Z","iopub.execute_input":"2024-05-27T07:34:10.775876Z","iopub.status.idle":"2024-05-27T07:34:11.069200Z","shell.execute_reply.started":"2024-05-27T07:34:10.775846Z","shell.execute_reply":"2024-05-27T07:34:11.068048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_resampled, y_resampled = oversample_minority(X, y)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:34:56.227517Z","iopub.execute_input":"2024-05-27T07:34:56.227933Z","iopub.status.idle":"2024-05-27T07:35:03.494093Z","shell.execute_reply.started":"2024-05-27T07:34:56.227903Z","shell.execute_reply":"2024-05-27T07:35:03.493127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_resampled.shape, y_resampled.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:35:03.495817Z","iopub.execute_input":"2024-05-27T07:35:03.496151Z","iopub.status.idle":"2024-05-27T07:35:03.503192Z","shell.execute_reply.started":"2024-05-27T07:35:03.496123Z","shell.execute_reply":"2024-05-27T07:35:03.501898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X_resampled\ny = y_resampled","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:35:03.504576Z","iopub.execute_input":"2024-05-27T07:35:03.504988Z","iopub.status.idle":"2024-05-27T07:35:03.515791Z","shell.execute_reply.started":"2024-05-27T07:35:03.504933Z","shell.execute_reply":"2024-05-27T07:35:03.514543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_resampled\ndel y_resampled\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:35:03.518189Z","iopub.execute_input":"2024-05-27T07:35:03.518536Z","iopub.status.idle":"2024-05-27T07:35:04.612282Z","shell.execute_reply.started":"2024-05-27T07:35:03.518508Z","shell.execute_reply":"2024-05-27T07:35:04.611240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weeks = X[\"WEEK_NUM\"]\ncase_id = X[\"case_id\"]\nX = X.drop(columns=[\"target\", \"case_id\",\"WEEK_NUM\"])\ny = y\n\n\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:35:04.613514Z","iopub.execute_input":"2024-05-27T07:35:04.613813Z","iopub.status.idle":"2024-05-27T07:35:07.627491Z","shell.execute_reply.started":"2024-05-27T07:35:04.613785Z","shell.execute_reply":"2024-05-27T07:35:07.626450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_1 = X\n#y_1 = y\n#weeks_1 = weeks","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:35:07.628734Z","iopub.execute_input":"2024-05-27T07:35:07.629057Z","iopub.status.idle":"2024-05-27T07:35:07.634093Z","shell.execute_reply.started":"2024-05-27T07:35:07.629031Z","shell.execute_reply":"2024-05-27T07:35:07.633028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nX = X.sample(10000, random_state=42)\ny = y.sample(10000, random_state=42)\nweeks = weeks.sample(10000, random_state=42)\n\"\"\"\n\n#X = X_1\n#y = y_1\n#weeks = weeks_1","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:40:20.218060Z","iopub.execute_input":"2024-05-27T07:40:20.219085Z","iopub.status.idle":"2024-05-27T07:40:20.228647Z","shell.execute_reply.started":"2024-05-27T07:40:20.219045Z","shell.execute_reply":"2024-05-27T07:40:20.227539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:38:45.915942Z","iopub.execute_input":"2024-05-27T07:38:45.916352Z","iopub.status.idle":"2024-05-27T07:38:45.927771Z","shell.execute_reply.started":"2024-05-27T07:38:45.916322Z","shell.execute_reply":"2024-05-27T07:38:45.926559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\n\nparams = {\n    \"device\":\"cpu\",\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 20,\n    \"learning_rate\": 0.05,\n    \"max_bin\": 255,\n    \"n_estimators\": 1200,\n    \"num_leaves\":30,\n    \"colsample_bytree\": 0.9, \n    \"colsample_bynode\": 0.9,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.01, \n    \"reg_lambda\": 3.25, \n    \"extra_trees\":True,\n    \n}\n\nfitted_models = []\ncv_scores = []  \n\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)\n\n#print(\"Valid week range: \", (weeks.iloc[idx_valid].min(), weeks.iloc[idx_valid].max()))\n\nmodel = lgb.LGBMClassifier(**params)\nmodel.fit(\n    X_train, y_train,\n    eval_set=[(X_valid, y_valid)],\n    callbacks=[lgb.log_evaluation(100), lgb.early_stopping(100)]\n)\n\nfitted_models.append(model)\n\ny_pred_valid = model.predict_proba(X_valid)[:, 1]\nauc_score = roc_auc_score(y_valid, y_pred_valid)\ncv_scores.append(auc_score)\n\nmodel = VotingModel(fitted_models)\nprint(\"CV AUC scores: \", cv_scores)\nprint(\"Average CV AUC score: \", sum(cv_scores) / len(cv_scores))","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:40:26.963846Z","iopub.execute_input":"2024-05-27T07:40:26.964623Z","iopub.status.idle":"2024-05-27T08:32:00.598023Z","shell.execute_reply.started":"2024-05-27T07:40:26.964588Z","shell.execute_reply":"2024-05-27T08:32:00.596541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"WEEK_NUM\"])\nX_test = X_test.set_index(\"case_id\")\n\nlgb_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)\nlgb_pred","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:39:14.516494Z","iopub.execute_input":"2024-05-27T07:39:14.516894Z","iopub.status.idle":"2024-05-27T07:39:14.547759Z","shell.execute_reply.started":"2024-05-27T07:39:14.516867Z","shell.execute_reply":"2024-05-27T07:39:14.546612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = lgb_pred","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:39:17.964996Z","iopub.execute_input":"2024-05-27T07:39:17.965781Z","iopub.status.idle":"2024-05-27T07:39:17.980459Z","shell.execute_reply.started":"2024-05-27T07:39:17.965747Z","shell.execute_reply":"2024-05-27T07:39:17.979434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:39:20.257962Z","iopub.execute_input":"2024-05-27T07:39:20.258514Z","iopub.status.idle":"2024-05-27T07:39:20.266323Z","shell.execute_reply.started":"2024-05-27T07:39:20.258473Z","shell.execute_reply":"2024-05-27T07:39:20.265018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:39:22.633771Z","iopub.execute_input":"2024-05-27T07:39:22.634589Z","iopub.status.idle":"2024-05-27T07:39:22.648532Z","shell.execute_reply.started":"2024-05-27T07:39:22.634551Z","shell.execute_reply":"2024-05-27T07:39:22.647479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())","metadata":{"execution":{"iopub.status.busy":"2024-05-27T08:51:04.869294Z","iopub.execute_input":"2024-05-27T08:51:04.870092Z","iopub.status.idle":"2024-05-27T08:51:04.881589Z","shell.execute_reply.started":"2024-05-27T08:51:04.870011Z","shell.execute_reply":"2024-05-27T08:51:04.880319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-27T08:51:18.885418Z","iopub.execute_input":"2024-05-27T08:51:18.885844Z","iopub.status.idle":"2024-05-27T08:51:18.899296Z","shell.execute_reply.started":"2024-05-27T08:51:18.885812Z","shell.execute_reply":"2024-05-27T08:51:18.898275Z"},"trusted":true},"execution_count":null,"outputs":[]}]}