{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":7584174,"sourceType":"datasetVersion","datasetId":4414761}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\n\nimport joblib\n\nimport lightgbm as lgb\nfrom imblearn.over_sampling import SMOTE\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8c8b5125-82e9-44e0-93de-d554dd2156b7","_cell_guid":"222536fc-204d-4cf3-a3f8-ce8d8accd51f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:17:47.944787Z","iopub.execute_input":"2024-02-14T00:17:47.945582Z","iopub.status.idle":"2024-02-14T00:17:54.507859Z","shell.execute_reply.started":"2024-02-14T00:17:47.945545Z","shell.execute_reply":"2024-02-14T00:17:54.506894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)","metadata":{"_uuid":"b63a63fc-22a1-4186-b452-bbd5e9cc3ff8","_cell_guid":"831347f7-4908-471c-a63d-9fc7e5eb5193","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:17:54.509797Z","iopub.execute_input":"2024-02-14T00:17:54.510523Z","iopub.status.idle":"2024-02-14T00:17:54.521207Z","shell.execute_reply.started":"2024-02-14T00:17:54.510464Z","shell.execute_reply":"2024-02-14T00:17:54.518886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"_uuid":"3726a6ae-1b74-4105-b9bc-3edb6e2063b2","_cell_guid":"45190116-64fe-4ef5-9f9b-6674b1ded3bb","collapsed":false,"execution":{"iopub.status.busy":"2024-02-14T00:17:54.526849Z","iopub.execute_input":"2024-02-14T00:17:54.527470Z","iopub.status.idle":"2024-02-14T00:17:54.563972Z","shell.execute_reply.started":"2024-02-14T00:17:54.527439Z","shell.execute_reply":"2024-02-14T00:17:54.562956Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"_uuid":"4d9469d9-1df1-4dea-bace-8007f6b35d6d","_cell_guid":"21c46c04-3baf-46d0-ace1-e9189d94c0e8","collapsed":false,"execution":{"iopub.status.busy":"2024-02-14T00:17:54.565283Z","iopub.execute_input":"2024-02-14T00:17:54.566273Z","iopub.status.idle":"2024-02-14T00:17:54.579014Z","shell.execute_reply.started":"2024-02-14T00:17:54.566238Z","shell.execute_reply":"2024-02-14T00:17:54.577973Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, float16_as32=True):\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage of dataframe is {start_mem:.2f} MB')\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type) == \"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    if float16_as32:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float16)                    \n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    \n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage after optimization is: {end_mem:.2f} MB')\n    print(f'Decreased by {(100 * (start_mem - end_mem) / start_mem):.1f}%')\n    \n    return df","metadata":{"_uuid":"8d4925a6-e59d-4f57-81e1-8528f083f030","_cell_guid":"6eecaba5-94bd-4cae-83bc-c0cc87b344cd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:17:54.580240Z","iopub.execute_input":"2024-02-14T00:17:54.580955Z","iopub.status.idle":"2024-02-14T00:17:54.598142Z","shell.execute_reply.started":"2024-02-14T00:17:54.580923Z","shell.execute_reply":"2024-02-14T00:17:54.596802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df","metadata":{"_uuid":"cead6108-034c-4a00-aad7-ac9c5069172a","_cell_guid":"66d65252-4f0a-4d46-8130-f6bdd0c91546","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:17:54.599113Z","iopub.execute_input":"2024-02-14T00:17:54.599388Z","iopub.status.idle":"2024-02-14T00:17:54.611423Z","shell.execute_reply.started":"2024-02-14T00:17:54.599363Z","shell.execute_reply":"2024-02-14T00:17:54.610612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"_uuid":"1dde9e42-187f-45ec-af3c-45a8bf222001","_cell_guid":"9e4d6ce5-49f8-4df1-ba9c-e8471fd3cc8e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:17:54.612583Z","iopub.execute_input":"2024-02-14T00:17:54.612888Z","iopub.status.idle":"2024-02-14T00:17:54.630567Z","shell.execute_reply.started":"2024-02-14T00:17:54.612865Z","shell.execute_reply":"2024-02-14T00:17:54.629642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"_uuid":"9e5a5c66-496c-4a7f-87a0-979cc9c3a319","_cell_guid":"49c5fb3e-e236-4204-91fc-1c7dbb2c33e6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:17:54.634128Z","iopub.execute_input":"2024-02-14T00:17:54.634436Z","iopub.status.idle":"2024-02-14T00:17:54.640976Z","shell.execute_reply.started":"2024-02-14T00:17:54.634408Z","shell.execute_reply":"2024-02-14T00:17:54.639928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"_uuid":"ff023c5f-ae0c-48d3-a7d4-a03c757c84e9","_cell_guid":"c41fedf1-0083-4270-aec0-e53d7af23257","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:17:54.644934Z","iopub.execute_input":"2024-02-14T00:17:54.645408Z","iopub.status.idle":"2024-02-14T00:17:54.651328Z","shell.execute_reply.started":"2024-02-14T00:17:54.645382Z","shell.execute_reply":"2024-02-14T00:17:54.650352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"_uuid":"08cdd90e-8e8b-4396-880b-6530662b9226","_cell_guid":"e1b9ae84-e385-4f3b-9326-67b339915b3d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:17:54.652293Z","iopub.execute_input":"2024-02-14T00:17:54.652543Z","iopub.status.idle":"2024-02-14T00:18:23.962992Z","shell.execute_reply.started":"2024-02-14T00:17:54.652523Z","shell.execute_reply":"2024-02-14T00:18:23.961967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"_uuid":"42cd79f9-e357-43a0-a47e-cf2aca887a03","_cell_guid":"99c9c215-9703-45c9-b51c-5dd7cbfbae2e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:18:23.964322Z","iopub.execute_input":"2024-02-14T00:18:23.964982Z","iopub.status.idle":"2024-02-14T00:18:31.096167Z","shell.execute_reply.started":"2024-02-14T00:18:23.964948Z","shell.execute_reply":"2024-02-14T00:18:31.095044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"_uuid":"c18e2d63-3e15-4b06-8a3c-aa3450e62d2f","_cell_guid":"529b6fa6-feaf-4088-9387-c74117ab36aa","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:18:31.097434Z","iopub.execute_input":"2024-02-14T00:18:31.097797Z","iopub.status.idle":"2024-02-14T00:18:31.469773Z","shell.execute_reply.started":"2024-02-14T00:18:31.097765Z","shell.execute_reply":"2024-02-14T00:18:31.468936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"_uuid":"d9b86b97-cfbf-4dd5-ac5e-896ced86d4c3","_cell_guid":"a2ca225c-fd34-4161-b5de-387d03e65fdd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:18:31.470966Z","iopub.execute_input":"2024-02-14T00:18:31.471332Z","iopub.status.idle":"2024-02-14T00:18:31.497923Z","shell.execute_reply.started":"2024-02-14T00:18:31.471300Z","shell.execute_reply":"2024-02-14T00:18:31.497065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"_uuid":"a0ad614d-e9bf-4afd-b18d-a44a100d70ac","_cell_guid":"347c1a4e-aaa1-4df8-9e42-2e5e40f62cb6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:18:31.499011Z","iopub.execute_input":"2024-02-14T00:18:31.499329Z","iopub.status.idle":"2024-02-14T00:18:33.642412Z","shell.execute_reply.started":"2024-02-14T00:18:31.499304Z","shell.execute_reply":"2024-02-14T00:18:33.641524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ndf_train = reduce_mem_usage(df_train)  # Optimize memory usage after conversion\n\ndf_test, cat_cols = to_pandas(df_test, cat_cols)\ndf_test = reduce_mem_usage(df_test)  # Optimize memory usage after conversion","metadata":{"_uuid":"b7e32954-14a8-4fbb-9884-795760e3b822","_cell_guid":"87b6bc72-21ea-4761-800d-8a89150a6e95","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:18:33.643690Z","iopub.execute_input":"2024-02-14T00:18:33.644046Z","iopub.status.idle":"2024-02-14T00:18:51.302881Z","shell.execute_reply.started":"2024-02-14T00:18:33.644013Z","shell.execute_reply":"2024-02-14T00:18:51.301851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\n\ngc.collect()","metadata":{"_uuid":"1465c91f-44d8-4922-b95e-b554bd0686c1","_cell_guid":"0148a009-aed9-47dc-a841-bc1a54871fb7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:18:51.303951Z","iopub.execute_input":"2024-02-14T00:18:51.304210Z","iopub.status.idle":"2024-02-14T00:18:51.440084Z","shell.execute_reply.started":"2024-02-14T00:18:51.304188Z","shell.execute_reply":"2024-02-14T00:18:51.439116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train is duplicated:\\t\", df_train[\"case_id\"].duplicated().any())\nprint(\"Train Week Range:\\t\", (df_train[\"WEEK_NUM\"].min(), df_train[\"WEEK_NUM\"].max()))\n\nprint()\n\nprint(\"Test is duplicated:\\t\", df_test[\"case_id\"].duplicated().any())\nprint(\"Test Week Range:\\t\", (df_test[\"WEEK_NUM\"].min(), df_test[\"WEEK_NUM\"].max()))","metadata":{"_uuid":"6501dc59-fbfd-41d1-b053-47ca6ff7d27c","_cell_guid":"cd545162-9fba-4d44-bdf4-4596ec5826aa","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:18:51.441299Z","iopub.execute_input":"2024-02-14T00:18:51.441600Z","iopub.status.idle":"2024-02-14T00:18:51.473803Z","shell.execute_reply.started":"2024-02-14T00:18:51.441575Z","shell.execute_reply":"2024-02-14T00:18:51.472855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(\n    data=df_train,\n    x=\"WEEK_NUM\",\n    y=\"target\",\n)\nplt.show()","metadata":{"_uuid":"be282851-e092-47ac-ae9f-1bc05c197f3f","_cell_guid":"ec75de88-ec4b-4cff-9e21-b09b3709e1ef","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:18:51.475036Z","iopub.execute_input":"2024-02-14T00:18:51.475287Z","iopub.status.idle":"2024-02-14T00:19:08.239515Z","shell.execute_reply.started":"2024-02-14T00:18:51.475265Z","shell.execute_reply":"2024-02-14T00:19:08.238560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\ny = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]\n\ncv = StratifiedGroupKFold(n_splits=3, shuffle=False)\n\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"n_estimators\": 10000,\n    \"colsample_bytree\": 0.8, \n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"device\": \"gpu\",\n    'max_depth': 16,  # Reduced from a larger value to control complexity\n    'min_data_in_leaf': 50,  # Increased from a lower value to control overfitting\n    'lambda_l1': 1.0,  # Regularization term\n    'lambda_l2': 1.0,  # Regularization term\n    'learning_rate': 0.001,\n    'num_leaves': 25,  # Reduce this if the model is overfitting\n    'bagging_freq': 5\n}\n\nfitted_models = []\n\nfor idx_train, idx_valid in cv.split(X, y, groups=weeks):\n    X_train, y_train = X.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = X.iloc[idx_valid], y.iloc[idx_valid]\n\n    model = lgb.LGBMClassifier(**params)\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        callbacks=[lgb.log_evaluation(100), lgb.early_stopping(100)]\n    )\n\n    fitted_models.append(model)\n\nmodel = VotingModel(fitted_models)","metadata":{"_uuid":"f2be2a98-83b5-4d11-beff-faffd230d886","_cell_guid":"80fb724c-2012-433a-a82d-99a743e48f57","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-14T00:19:08.241018Z","iopub.execute_input":"2024-02-14T00:19:08.241396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"WEEK_NUM\"])\nX_test = X_test.set_index(\"case_id\")\n\ny_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)\n\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred\n\nprint(\"Check null: \", df_subm[\"score\"].isnull().any())\n\ndf_subm.head()\n\ndf_subm.to_csv(\"submission.csv\")","metadata":{"_uuid":"a8254857-39f6-41af-bd55-ced30f3fcdc1","_cell_guid":"b5e01941-2f96-4190-8b0c-afd1197b7c99","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}