{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7894659,"sourceType":"datasetVersion","datasetId":4635391},{"sourceId":162314401,"sourceType":"kernelVersion"},{"sourceId":162317063,"sourceType":"kernelVersion"},{"sourceId":162351144,"sourceType":"kernelVersion"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Resource\n- training notebook\n  - https://www.kaggle.com/code/motono0223/home-credit-automl-training\n- packages for offline installation\n  - https://www.kaggle.com/code/motono0223/autogluon-pkgs\n  - https://www.kaggle.com/code/motono0223/ray-pkgs\n  \n# Reference \n- [1] [home-credit-baseline](https://www.kaggle.com/code/greysky/home-credit-baseline)\n- [2] [home-credit-baseline-max-min-features](https://www.kaggle.com/code/stechparme/home-credit-baseline-max-min-features)\n- [3] [dependency of autogluon (version confliction by ray package)](https://github.com/autogluon/autogluon/issues/3365)\n- [4] [Autogluon APIs](https://auto.gluon.ai/stable/api/autogluon.tabular.TabularPredictor.html)","metadata":{}},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/autogluon-pkgs autogluon > /dev/null","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-03-20T10:44:29.011010Z","iopub.execute_input":"2024-03-20T10:44:29.012010Z","iopub.status.idle":"2024-03-20T10:44:52.263187Z","shell.execute_reply.started":"2024-03-20T10:44:29.011965Z","shell.execute_reply":"2024-03-20T10:44:52.261957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/ray-pkgs --upgrade --force-reinstall -q ray==2.6.3","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-03-20T10:44:52.269581Z","iopub.execute_input":"2024-03-20T10:44:52.269935Z","iopub.status.idle":"2024-03-20T10:45:17.454618Z","shell.execute_reply.started":"2024-03-20T10:44:52.269889Z","shell.execute_reply":"2024-03-20T10:45:17.453354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom autogluon.tabular import TabularDataset, TabularPredictor","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:17.456866Z","iopub.execute_input":"2024-03-20T10:45:17.457275Z","iopub.status.idle":"2024-03-20T10:45:19.878327Z","shell.execute_reply.started":"2024-03-20T10:45:17.457236Z","shell.execute_reply":"2024-03-20T10:45:19.877118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:19.881128Z","iopub.execute_input":"2024-03-20T10:45:19.881798Z","iopub.status.idle":"2024-03-20T10:45:20.021397Z","shell.execute_reply.started":"2024-03-20T10:45:19.881735Z","shell.execute_reply":"2024-03-20T10:45:20.020077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline","metadata":{}},{"cell_type":"markdown","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df): #Standardize the dtype.\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df): #Change the feature for D to the difference in days from date_decision.\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df): #Remove those with an average is_null exceeding 0.95 and those that do not fall within the range 1 < nunique < 200.\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-03-20T06:33:02.742685Z","iopub.execute_input":"2024-03-20T06:33:02.743431Z","iopub.status.idle":"2024-03-20T06:33:02.759908Z","shell.execute_reply.started":"2024-03-20T06:33:02.743392Z","shell.execute_reply":"2024-03-20T06:33:02.758691Z"}}},{"cell_type":"markdown","source":"# Automatic Aggregation","metadata":{}},{"cell_type":"markdown","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df): #Extract the maximum and minimum values for features P and A, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def date_expr(df): #Extract the maximum and minimum values for features D, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def str_expr(df): #Extract the maximum and minimum values for features M, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def other_expr(df): #Extract the maximum and minimum values for features T and L, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n    \n    @staticmethod\n    def count_expr(df): #Extract the maximum and minimum values for each num_group and add them as additional features.\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def get_exprs(df): #Execute the above function and return the result.\n        maxexprs = Aggregator.num_expr(df)[0] + \\\n                Aggregator.date_expr(df)[0] + \\\n                Aggregator.str_expr(df)[0] + \\\n                Aggregator.other_expr(df)[0] + \\\n                Aggregator.count_expr(df)[0]\n        \n        minexprs = Aggregator.num_expr(df)[1] + \\\n                Aggregator.date_expr(df)[1] + \\\n                Aggregator.str_expr(df)[1] + \\\n                Aggregator.other_expr(df)[1] + \\\n                Aggregator.count_expr(df)[1]\n\n        return maxexprs, minexprs","metadata":{"execution":{"iopub.status.busy":"2024-03-20T06:33:02.761504Z","iopub.execute_input":"2024-03-20T06:33:02.761931Z","iopub.status.idle":"2024-03-20T06:33:02.784959Z","shell.execute_reply.started":"2024-03-20T06:33:02.761897Z","shell.execute_reply":"2024-03-20T06:33:02.782713Z"}}},{"cell_type":"code","source":"\n    \nclass Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df, train_cols = []):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                \n                # do not drop column if it was part of train column\n                if col in train_cols:\n                    continue\n                \n                \n                isnull = df[col].is_null().mean()\n                \n                # if majority is NULL drop column\n                if isnull > 0.8: #0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                \n                # do not drop column if it was part of train column\n                if col in train_cols:\n                    continue\n                \n                freq = df[col].n_unique()\n                \n                # if too many unique values or single value\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df\n    \nclass Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs\n","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.022919Z","iopub.execute_input":"2024-03-20T10:45:20.023712Z","iopub.status.idle":"2024-03-20T10:45:20.153436Z","shell.execute_reply.started":"2024-03-20T10:45:20.023656Z","shell.execute_reply":"2024-03-20T10:45:20.152257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File I/O","metadata":{}},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob.glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.154777Z","iopub.execute_input":"2024-03-20T10:45:20.155397Z","iopub.status.idle":"2024-03-20T10:45:20.171536Z","shell.execute_reply.started":"2024-03-20T10:45:20.155362Z","shell.execute_reply":"2024-03-20T10:45:20.170187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"def read_file(path, depth=None): \n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        maxexprs, minexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs)\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob.glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        maxexprs, minexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-20T06:33:02.786743Z","iopub.execute_input":"2024-03-20T06:33:02.787217Z","iopub.status.idle":"2024-03-20T06:33:02.800174Z","shell.execute_reply.started":"2024-03-20T06:33:02.787182Z","shell.execute_reply":"2024-03-20T06:33:02.798830Z"}}},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.173190Z","iopub.execute_input":"2024-03-20T10:45:20.174060Z","iopub.status.idle":"2024-03-20T10:45:20.186678Z","shell.execute_reply.started":"2024-03-20T10:45:20.174010Z","shell.execute_reply":"2024-03-20T10:45:20.185816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.188362Z","iopub.execute_input":"2024-03-20T10:45:20.188736Z","iopub.status.idle":"2024-03-20T10:45:20.199001Z","shell.execute_reply.started":"2024-03-20T10:45:20.188702Z","shell.execute_reply":"2024-03-20T10:45:20.197840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.201015Z","iopub.execute_input":"2024-03-20T10:45:20.201567Z","iopub.status.idle":"2024-03-20T10:45:20.214890Z","shell.execute_reply.started":"2024-03-20T10:45:20.201528Z","shell.execute_reply":"2024-03-20T10:45:20.213679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"TRAINING = False","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.216811Z","iopub.execute_input":"2024-03-20T10:45:20.217616Z","iopub.status.idle":"2024-03-20T10:45:20.230675Z","shell.execute_reply.started":"2024-03-20T10:45:20.217567Z","shell.execute_reply":"2024-03-20T10:45:20.229442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\n#DRY_RUN = True if sample.shape[0] == 10 else False   # if num of records of test data is 10, dry-run is enable.\n\nMODEL_PATH = \"/kaggle/input/homecredit-models/mymodel_v1/my_models_2\"\n","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.232317Z","iopub.execute_input":"2024-03-20T10:45:20.233613Z","iopub.status.idle":"2024-03-20T10:45:20.248708Z","shell.execute_reply.started":"2024-03-20T10:45:20.233561Z","shell.execute_reply":"2024-03-20T10:45:20.247623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.250901Z","iopub.execute_input":"2024-03-20T10:45:20.251933Z","iopub.status.idle":"2024-03-20T10:45:20.261535Z","shell.execute_reply.started":"2024-03-20T10:45:20.251881Z","shell.execute_reply":"2024-03-20T10:45:20.260152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"def get_train_data_store():\n\n    data_store = {\n        \"df_base\": read_file(os.path.join(TRAIN_DIR , \"train_base.parquet\")),\n        \"depth_0\": [\n            read_file(os.path.join(TRAIN_DIR , \"train_static_cb_0.parquet\")),\n            read_files(os.path.join(TRAIN_DIR , \"train_static_0_*.parquet\")),\n        ],\n        \"depth_1\": [\n            read_files(os.path.join(TRAIN_DIR , \"train_applprev_1_*.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_tax_registry_a_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_tax_registry_b_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_tax_registry_c_1.parquet\"), 1),\n            read_files(os.path.join(TRAIN_DIR , \"train_credit_bureau_a_1_*.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_credit_bureau_b_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_other_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_person_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_deposit_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_debitcard_1.parquet\"), 1),\n        ],\n        \"depth_2\": [\n            read_file(os.path.join(TRAIN_DIR , \"train_credit_bureau_b_2.parquet\"), 2),\n            read_files(os.path.join(TRAIN_DIR , \"train_credit_bureau_a_2_*.parquet\"), 2),\n        ]\n    }\n\n    return data_store\n\ndef get_test_data_store():\n    data_store = {\n        \"df_base\": read_file(os.path.join(TEST_DIR , \"test_base.parquet\")),\n        \"depth_0\": [\n            read_file(os.path.join(TEST_DIR , \"test_static_cb_0.parquet\")),\n            read_files(os.path.join(TEST_DIR , \"test_static_0_*.parquet\")),\n        ],\n        \"depth_1\": [\n            read_files(os.path.join(TEST_DIR, \"test_applprev_1_*.parquet\"), 1),\n            read_file(os.path.join(TEST_DIR , \"test_tax_registry_a_1.parquet\"), 1),\n            read_file(os.path.join(TEST_DIR , \"test_tax_registry_b_1.parquet\"), 1),\n            read_file(os.path.join(TEST_DIR , \"test_tax_registry_c_1.parquet\"), 1),\n            read_files(os.path.join(TEST_DIR, \"test_credit_bureau_a_1_*.parquet\"), 1),\n            read_file(os.path.join(TEST_DIR , \"test_credit_bureau_b_1.parquet\"), 1),\n            read_file(os.path.join(TEST_DIR , \"test_other_1.parquet\"), 1),\n            read_file(os.path.join(TEST_DIR , \"test_person_1.parquet\"), 1),\n            read_file(os.path.join(TEST_DIR , \"test_deposit_1.parquet\"), 1),\n            read_file(os.path.join(TEST_DIR , \"test_debitcard_1.parquet\"), 1),\n        ],\n        \"depth_2\": [\n            read_file(os.path.join(TEST_DIR , \"test_credit_bureau_b_2.parquet\"), 2),\n            read_files(os.path.join(TEST_DIR , \"test_credit_bureau_a_2_*.parquet\"), 2),\n        ]\n    }\n    return data_store","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.269123Z","iopub.execute_input":"2024-03-20T10:45:20.269633Z","iopub.status.idle":"2024-03-20T10:45:20.282695Z","shell.execute_reply.started":"2024-03-20T10:45:20.269597Z","shell.execute_reply":"2024-03-20T10:45:20.281614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del data_store\n#gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.285039Z","iopub.execute_input":"2024-03-20T10:45:20.285617Z","iopub.status.idle":"2024-03-20T10:45:20.304449Z","shell.execute_reply.started":"2024-03-20T10:45:20.285566Z","shell.execute_reply":"2024-03-20T10:45:20.302123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-03-20T06:37:32.961128Z","iopub.execute_input":"2024-03-20T06:37:32.961564Z"}}},{"cell_type":"markdown","source":"data_store = {\n        \"df_base\": read_file(os.path.join(TRAIN_DIR , \"train_base.parquet\")),\n        \"depth_0\": [\n            read_file(os.path.join(TRAIN_DIR , \"train_static_cb_0.parquet\")),\n            read_files(os.path.join(TRAIN_DIR , \"train_static_0_*.parquet\")),\n        ],\n        \"depth_1\": [\n            read_files(os.path.join(TRAIN_DIR , \"train_applprev_1_*.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_tax_registry_a_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_tax_registry_b_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_tax_registry_c_1.parquet\"), 1),\n            read_files(os.path.join(TRAIN_DIR , \"train_credit_bureau_a_1_*.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_credit_bureau_b_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_other_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_person_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_deposit_1.parquet\"), 1),\n            read_file(os.path.join(TRAIN_DIR , \"train_debitcard_1.parquet\"), 1),\n        ],\n        \"depth_2\": [\n            read_file(os.path.join(TRAIN_DIR , \"train_credit_bureau_b_2.parquet\"), 2),\n            read_files(os.path.join(TRAIN_DIR , \"train_credit_bureau_a_2_*.parquet\"), 2),\n        ]\n    }","metadata":{"execution":{"iopub.status.busy":"2024-03-20T06:28:41.125272Z","iopub.execute_input":"2024-03-20T06:28:41.125811Z"}}},{"cell_type":"code","source":"if TRAINING:\n    data_store = get_train_data_store()\n    len(data_store)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.308422Z","iopub.execute_input":"2024-03-20T10:45:20.309469Z","iopub.status.idle":"2024-03-20T10:45:20.317720Z","shell.execute_reply.started":"2024-03-20T10:45:20.309416Z","shell.execute_reply":"2024-03-20T10:45:20.316481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    df_train = feature_eng(**data_store)\n    print(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.319248Z","iopub.execute_input":"2024-03-20T10:45:20.319739Z","iopub.status.idle":"2024-03-20T10:45:20.329767Z","shell.execute_reply.started":"2024-03-20T10:45:20.319676Z","shell.execute_reply":"2024-03-20T10:45:20.328799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    df_train = df_train.pipe(Pipeline.filter_cols)\n    print(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.331573Z","iopub.execute_input":"2024-03-20T10:45:20.332685Z","iopub.status.idle":"2024-03-20T10:45:20.342609Z","shell.execute_reply.started":"2024-03-20T10:45:20.332645Z","shell.execute_reply":"2024-03-20T10:45:20.341551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    df_train, cat_cols = to_pandas(df_train)\n    print(df_train.shape)\n    df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.344117Z","iopub.execute_input":"2024-03-20T10:45:20.344715Z","iopub.status.idle":"2024-03-20T10:45:20.354540Z","shell.execute_reply.started":"2024-03-20T10:45:20.344680Z","shell.execute_reply":"2024-03-20T10:45:20.353535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    del data_store\n    df_train = reduce_mem_usage(df_train)\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.356059Z","iopub.execute_input":"2024-03-20T10:45:20.356606Z","iopub.status.idle":"2024-03-20T10:45:20.364806Z","shell.execute_reply.started":"2024-03-20T10:45:20.356574Z","shell.execute_reply":"2024-03-20T10:45:20.363888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:47:55.531793Z","iopub.execute_input":"2024-03-20T10:47:55.532757Z","iopub.status.idle":"2024-03-20T10:47:55.538645Z","shell.execute_reply.started":"2024-03-20T10:47:55.532696Z","shell.execute_reply":"2024-03-20T10:47:55.537281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    train_info = {\n        'train_cols': train_cols,\n        'cat_cols': cat_cols\n    }\n    train_cols = df_train.columns\n    with open('train_info.pkl', 'wb+') as f:\n        pickle.dump(train_info, f)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:47:56.795130Z","iopub.execute_input":"2024-03-20T10:47:56.795888Z","iopub.status.idle":"2024-03-20T10:47:56.801098Z","shell.execute_reply.started":"2024-03-20T10:47:56.795846Z","shell.execute_reply":"2024-03-20T10:47:56.800044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#rain_cols = train_cols.columns\n#train_cols","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:47:58.369229Z","iopub.execute_input":"2024-03-20T10:47:58.369646Z","iopub.status.idle":"2024-03-20T10:47:58.374168Z","shell.execute_reply.started":"2024-03-20T10:47:58.369616Z","shell.execute_reply":"2024-03-20T10:47:58.372822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    predictor = TabularPredictor(\n        label=\"target\",\n        problem_type=\"binary\",\n        eval_metric=\"roc_auc\",\n        path=\"my_models_2\",\n    )\n\n# Load model\n#predictor = TabularPredictor.load(path=MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:47:59.454689Z","iopub.execute_input":"2024-03-20T10:47:59.455169Z","iopub.status.idle":"2024-03-20T10:47:59.461398Z","shell.execute_reply.started":"2024-03-20T10:47:59.455128Z","shell.execute_reply":"2024-03-20T10:47:59.460014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    weeks = df_train[\"WEEK_NUM\"]\n    df_train = df_train.drop(columns=[\"case_id\", \"WEEK_NUM\"])\n    cv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n    for idx_train, idx_valid in cv.split(df_train, df_train[\"target\"], groups=weeks):\n        fold_train = df_train.iloc[idx_train]\n        fold_valid = df_train.iloc[idx_valid]\n        train_data = TabularDataset(fold_train)\n        valid_data = TabularDataset(fold_valid)\n        break","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:47:59.808308Z","iopub.execute_input":"2024-03-20T10:47:59.808983Z","iopub.status.idle":"2024-03-20T10:47:59.815052Z","shell.execute_reply.started":"2024-03-20T10:47:59.808947Z","shell.execute_reply":"2024-03-20T10:47:59.813856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    del df_train # It is deleted to reduce memory usage\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:00.960623Z","iopub.execute_input":"2024-03-20T10:48:00.961220Z","iopub.status.idle":"2024-03-20T10:48:00.968462Z","shell.execute_reply.started":"2024-03-20T10:48:00.961171Z","shell.execute_reply":"2024-03-20T10:48:00.966520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nif TRAINING:\n    PRESETS =  \"optimize_for_deployment\" #\"medium_quality\"#\"best_quality\", \"high_quality\", \"good_quality\", \"medium_quality\", \"optimize_for_deployment\"\n\n    predictor.fit(\n        train_data,\n        tuning_data=valid_data,\n        save_space=True,\n        presets=PRESETS,\n\n        use_bag_holdout=True,\n        #ag_args_fit={'num_gpus': 1},\n    )","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-03-20T10:48:01.410027Z","iopub.execute_input":"2024-03-20T10:48:01.411144Z","iopub.status.idle":"2024-03-20T10:48:01.421304Z","shell.execute_reply.started":"2024-03-20T10:48:01.411078Z","shell.execute_reply":"2024-03-20T10:48:01.419980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:02.594980Z","iopub.execute_input":"2024-03-20T10:48:02.596392Z","iopub.status.idle":"2024-03-20T10:48:02.601822Z","shell.execute_reply.started":"2024-03-20T10:48:02.596341Z","shell.execute_reply":"2024-03-20T10:48:02.600068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Useless Original Features (Count: 1):\n        ['deferredmnthsnum_166L']\n\t\tThese features carry no predictive signal and should be manually investigated.\n\t\tThis is typically a feature which has the same value for all rows.\n\t\tThese features do not need to be present at inference time.\n\t- Unused Original Features (Count: 15): \n    ['numberofqueries_373L', 'inittransactioncode_186L', 'interestrate_311L', 'mastercontrexist_109L', 'paytype_783L', 'max_approvaldate_319D', 'max_creationdate_885D', 'max_dateactivated_425D', 'max_tenor_203L', 'max_empladdr_zipcode_114M', 'max_persontype_792L', 'max_relationshiptoclient_642T', 'max_pmts_overdue_1140A', 'max_pmts_overdue_1152A', 'max_collater_typofvalofguarant_407M']\n\t- \tThese features were not used to generate any of the output features. Add a feature generator compatible with these features to utilize them.\n\t\tFeatures can also be unused if they carry very little information, such as being categorical but having almost entirely unique values or being duplicates of other features.\n\t- \tThese features do not need to be present at inference time.\n\t\t('category', []) :  5 | \n        ['inittransactioncode_186L', 'paytype_783L', 'max_empladdr_zipcode_114M', 'max_relationshiptoclient_642T', 'max_collater_typofvalofguarant_407M']\n        \n\t\t('float', [])    : 10 | \n        ['numberofqueries_373L', 'interestrate_311L', 'mastercontrexist_109L', 'max_approvaldate_319D', 'max_creationdate_885D', ...]","metadata":{}},{"cell_type":"markdown","source":"{\n\t'NN_TORCH': [{}, {'activation': 'elu', 'dropout_prob': 0.10077639529843717, 'hidden_size': 108, 'learning_rate': 0.002735937344002146, 'num_layers': 4, 'use_batchnorm': True, 'weight_decay': 1.356433327634438e-12, 'ag_args': {'name_suffix': '_r79', 'priority': -2}}, {'activation': 'elu', 'dropout_prob': 0.11897478034205347, 'hidden_size': 213, 'learning_rate': 0.0010474382260641949, 'num_layers': 4, 'use_batchnorm': False, 'weight_decay': 5.594471067786272e-10, 'ag_args': {'name_suffix': '_r22', 'priority': -7}}],\n\t'GBM': [{'extra_trees': True, 'ag_args': {'name_suffix': 'XT'}}, {}, 'GBMLarge'],\n\t'CAT': [{}, {'depth': 6, 'grow_policy': 'SymmetricTree', 'l2_leaf_reg': 2.1542798306067823, 'learning_rate': 0.06864209415792857, 'max_ctr_complexity': 4, 'one_hot_max_size': 10, 'ag_args': {'name_suffix': '_r177', 'priority': -1}}, {'depth': 8, 'grow_policy': 'Depthwise', 'l2_leaf_reg': 2.7997999596449104, 'learning_rate': 0.031375015734637225, 'max_ctr_complexity': 2, 'one_hot_max_size': 3, 'ag_args': {'name_suffix': '_r9', 'priority': -5}}],\n\t'XGB': [{}, {'colsample_bytree': 0.6917311125174739, 'enable_categorical': False, 'learning_rate': 0.018063876087523967, 'max_depth': 10, 'min_child_weight': 0.6028633586934382, 'ag_args': {'name_suffix': '_r33', 'priority': -8}}, {'colsample_bytree': 0.6628423832084077, 'enable_categorical': False, 'learning_rate': 0.08775715546881824, 'max_depth': 5, 'min_child_weight': 0.6294123374222513, 'ag_args': {'name_suffix': '_r89', 'priority': -16}}],\n\t'FASTAI': [{}, {'bs': 256, 'emb_drop': 0.5411770367537934, 'epochs': 43, 'layers': [800, 400], 'lr': 0.01519848858318159, 'ps': 0.23782946566604385, 'ag_args': {'name_suffix': '_r191', 'priority': -4}}, {'bs': 2048, 'emb_drop': 0.05070411322605811, 'epochs': 29, 'layers': [200, 100], 'lr': 0.08974235041576624, 'ps': 0.10393466140748028, 'ag_args': {'name_suffix': '_r102', 'priority': -11}}],\n\t'RF': [{'criterion': 'gini', 'ag_args': {'name_suffix': 'Gini', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'entropy', 'ag_args': {'name_suffix': 'Entr', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'squared_error', 'ag_args': {'name_suffix': 'MSE', 'problem_types': ['regression', 'quantile']}}],\n\t'XT': [{'criterion': 'gini', 'ag_args': {'name_suffix': 'Gini', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'entropy', 'ag_args': {'name_suffix': 'Entr', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'squared_error', 'ag_args': {'name_suffix': 'MSE', 'problem_types': ['regression', 'quantile']}}],\n\t'KNN': [{'weights': 'uniform', 'ag_args': {'name_suffix': 'Unif'}}, {'weights': 'distance', 'ag_args': {'name_suffix': 'Dist'}}],\n}","metadata":{}},{"cell_type":"markdown","source":"# Training result","metadata":{}},{"cell_type":"code","source":"# Load model\n#MODEL_PATH = '/kaggle/input/homecredit-models/results/my_models_2'\npredictor = TabularPredictor.load(path=MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:04.600159Z","iopub.execute_input":"2024-03-20T10:48:04.600619Z","iopub.status.idle":"2024-03-20T10:48:04.623682Z","shell.execute_reply.started":"2024-03-20T10:48:04.600581Z","shell.execute_reply":"2024-03-20T10:48:04.622688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictor2.leaderboard()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:06.994569Z","iopub.execute_input":"2024-03-20T10:48:06.995012Z","iopub.status.idle":"2024-03-20T10:48:07.000373Z","shell.execute_reply.started":"2024-03-20T10:48:06.994979Z","shell.execute_reply":"2024-03-20T10:48:06.998851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictor.leaderboard()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:07.393184Z","iopub.execute_input":"2024-03-20T10:48:07.393622Z","iopub.status.idle":"2024-03-20T10:48:07.423855Z","shell.execute_reply.started":"2024-03-20T10:48:07.393585Z","shell.execute_reply":"2024-03-20T10:48:07.422619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_lb = predictor.leaderboard()\nfrom matplotlib import pyplot as plt\nplt.scatter( df_lb[\"score_val\"], df_lb[\"model\"] )\nplt.grid()\nplt.xlabel(\"CV(roc_auc)\")\nplt.ylabel(\"Model name\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-20T10:48:10.985010Z","iopub.execute_input":"2024-03-20T10:48:10.985458Z","iopub.status.idle":"2024-03-20T10:48:11.248567Z","shell.execute_reply.started":"2024-03-20T10:48:10.985422Z","shell.execute_reply":"2024-03-20T10:48:11.247078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = get_test_data_store()\ndf_test = feature_eng(**data_store)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:15.144386Z","iopub.execute_input":"2024-03-20T10:48:15.144855Z","iopub.status.idle":"2024-03-20T10:48:15.401354Z","shell.execute_reply.started":"2024-03-20T10:48:15.144817Z","shell.execute_reply":"2024-03-20T10:48:15.399912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PATH","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:15.403582Z","iopub.execute_input":"2024-03-20T10:48:15.403951Z","iopub.status.idle":"2024-03-20T10:48:15.409696Z","shell.execute_reply.started":"2024-03-20T10:48:15.403921Z","shell.execute_reply":"2024-03-20T10:48:15.408548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#with open(os.path.join(MODEL_PATH, 'train_info.pkl'), 'rb') as f:\n#        train_info = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:15.452726Z","iopub.execute_input":"2024-03-20T10:48:15.453437Z","iopub.status.idle":"2024-03-20T10:48:15.457491Z","shell.execute_reply.started":"2024-03-20T10:48:15.453401Z","shell.execute_reply":"2024-03-20T10:48:15.456207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_info","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:15.679296Z","iopub.execute_input":"2024-03-20T10:48:15.679759Z","iopub.status.idle":"2024-03-20T10:48:15.684897Z","shell.execute_reply.started":"2024-03-20T10:48:15.679705Z","shell.execute_reply":"2024-03-20T10:48:15.683672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_cols = train_info['train_cols']\n#cat_cols = train_info['cat_cols']","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:17.297704Z","iopub.execute_input":"2024-03-20T10:48:17.298210Z","iopub.status.idle":"2024-03-20T10:48:17.303515Z","shell.execute_reply.started":"2024-03-20T10:48:17.298161Z","shell.execute_reply":"2024-03-20T10:48:17.302207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAINING","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:17.519526Z","iopub.execute_input":"2024-03-20T10:48:17.519988Z","iopub.status.idle":"2024-03-20T10:48:17.528021Z","shell.execute_reply.started":"2024-03-20T10:48:17.519950Z","shell.execute_reply":"2024-03-20T10:48:17.526508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not TRAINING:\n    with open(os.path.join(MODEL_PATH, 'train_info.pkl'), 'rb') as f:\n        train_info = pickle.load(f)\n     \n    train_cols = train_info['train_cols']\n    cat_cols = train_info['cat_cols']","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:19.544317Z","iopub.execute_input":"2024-03-20T10:48:19.544797Z","iopub.status.idle":"2024-03-20T10:48:19.551277Z","shell.execute_reply.started":"2024-03-20T10:48:19.544757Z","shell.execute_reply":"2024-03-20T10:48:19.550364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_cols","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:19.994690Z","iopub.execute_input":"2024-03-20T10:48:19.995736Z","iopub.status.idle":"2024-03-20T10:48:20.000587Z","shell.execute_reply.started":"2024-03-20T10:48:19.995689Z","shell.execute_reply":"2024-03-20T10:48:19.999397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_test = df_test.select([col for col in df_test.columns if col != \"target\"])\ndf_test = df_test.select([col for col in train_cols if col != \"target\"])\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:20.465435Z","iopub.execute_input":"2024-03-20T10:48:20.465884Z","iopub.status.idle":"2024-03-20T10:48:20.475793Z","shell.execute_reply.started":"2024-03-20T10:48:20.465849Z","shell.execute_reply":"2024-03-20T10:48:20.474117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test, cat_cols = to_pandas(df_test, cat_cols) # cat_cols was created by train data","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:21.869455Z","iopub.execute_input":"2024-03-20T10:48:21.869936Z","iopub.status.idle":"2024-03-20T10:48:21.939088Z","shell.execute_reply.started":"2024-03-20T10:48:21.869897Z","shell.execute_reply":"2024-03-20T10:48:21.937987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_test.shape)\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:22.179338Z","iopub.execute_input":"2024-03-20T10:48:22.180703Z","iopub.status.idle":"2024-03-20T10:48:22.214112Z","shell.execute_reply.started":"2024-03-20T10:48:22.180651Z","shell.execute_reply":"2024-03-20T10:48:22.212805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ndf_test = reduce_mem_usage(df_test)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:22.499251Z","iopub.execute_input":"2024-03-20T10:48:22.499709Z","iopub.status.idle":"2024-03-20T10:48:22.809136Z","shell.execute_reply.started":"2024-03-20T10:48:22.499669Z","shell.execute_reply":"2024-03-20T10:48:22.807848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"case_id\", \"WEEK_NUM\"])\ntest_data = TabularDataset(df_test)\ny_pred = predictor.predict_proba(test_data).iloc[:, 1].values","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:24.615039Z","iopub.execute_input":"2024-03-20T10:48:24.615940Z","iopub.status.idle":"2024-03-20T10:48:24.870115Z","shell.execute_reply.started":"2024-03-20T10:48:24.615886Z","shell.execute_reply":"2024-03-20T10:48:24.869077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:26.259503Z","iopub.execute_input":"2024-03-20T10:48:26.260176Z","iopub.status.idle":"2024-03-20T10:48:26.269196Z","shell.execute_reply.started":"2024-03-20T10:48:26.260140Z","shell.execute_reply":"2024-03-20T10:48:26.267965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_subm","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:27.693508Z","iopub.execute_input":"2024-03-20T10:48:27.694044Z","iopub.status.idle":"2024-03-20T10:48:27.698442Z","shell.execute_reply.started":"2024-03-20T10:48:27.694005Z","shell.execute_reply":"2024-03-20T10:48:27.697349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())\n\ndf_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:27.994806Z","iopub.execute_input":"2024-03-20T10:48:27.995479Z","iopub.status.idle":"2024-03-20T10:48:28.009142Z","shell.execute_reply.started":"2024-03-20T10:48:27.995443Z","shell.execute_reply":"2024-03-20T10:48:28.007924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:48:28.974667Z","iopub.execute_input":"2024-03-20T10:48:28.975131Z","iopub.status.idle":"2024-03-20T10:48:28.983642Z","shell.execute_reply.started":"2024-03-20T10:48:28.975096Z","shell.execute_reply":"2024-03-20T10:48:28.982226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!cd /kaggle/working\n#!tar -czvf my_work.zip -C . .","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.860956Z","iopub.status.idle":"2024-03-20T10:45:20.861302Z","shell.execute_reply.started":"2024-03-20T10:45:20.861123Z","shell.execute_reply":"2024-03-20T10:45:20.861138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from IPython.display import FileLink","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.862625Z","iopub.status.idle":"2024-03-20T10:45:20.863010Z","shell.execute_reply.started":"2024-03-20T10:45:20.862831Z","shell.execute_reply":"2024-03-20T10:45:20.862848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FileLink('/kaggle/working/my_work.zip')","metadata":{"execution":{"iopub.status.busy":"2024-03-20T10:45:20.864046Z","iopub.status.idle":"2024-03-20T10:45:20.864403Z","shell.execute_reply.started":"2024-03-20T10:45:20.864231Z","shell.execute_reply":"2024-03-20T10:45:20.864247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}