{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":162314401,"sourceType":"kernelVersion"},{"sourceId":162317063,"sourceType":"kernelVersion"},{"sourceId":162351144,"sourceType":"kernelVersion"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Resource\n- training notebook\n  - https://www.kaggle.com/code/motono0223/home-credit-automl-training\n- packages for offline installation\n  - https://www.kaggle.com/code/motono0223/autogluon-pkgs\n  - https://www.kaggle.com/code/motono0223/ray-pkgs\n  \n# Reference \n- [1] [home-credit-baseline](https://www.kaggle.com/code/greysky/home-credit-baseline)\n- [2] [home-credit-baseline-max-min-features](https://www.kaggle.com/code/stechparme/home-credit-baseline-max-min-features)\n- [3] [dependency of autogluon (version confliction by ray package)](https://github.com/autogluon/autogluon/issues/3365)\n- [4] [Autogluon APIs](https://auto.gluon.ai/stable/api/autogluon.tabular.TabularPredictor.html)","metadata":{}},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/autogluon-pkgs autogluon > /dev/null","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-02T09:05:43.689789Z","iopub.execute_input":"2024-04-02T09:05:43.690177Z","iopub.status.idle":"2024-04-02T09:06:06.901714Z","shell.execute_reply.started":"2024-04-02T09:05:43.690149Z","shell.execute_reply":"2024-04-02T09:06:06.900470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/ray-pkgs --upgrade --force-reinstall -q ray==2.6.3","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-02T09:06:06.904090Z","iopub.execute_input":"2024-04-02T09:06:06.904449Z","iopub.status.idle":"2024-04-02T09:06:31.926141Z","shell.execute_reply.started":"2024-04-02T09:06:06.904420Z","shell.execute_reply":"2024-04-02T09:06:31.925056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom autogluon.tabular import TabularDataset, TabularPredictor","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:31.927616Z","iopub.execute_input":"2024-04-02T09:06:31.927944Z","iopub.status.idle":"2024-04-02T09:06:31.935492Z","shell.execute_reply.started":"2024-04-02T09:06:31.927916Z","shell.execute_reply":"2024-04-02T09:06:31.934420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df): #Standardize the dtype.\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df): #Change the feature for D to the difference in days from date_decision.\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df): #Remove those with an average is_null exceeding 0.95 and those that do not fall within the range 1 < nunique < 200.\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:31.938782Z","iopub.execute_input":"2024-04-02T09:06:31.939097Z","iopub.status.idle":"2024-04-02T09:06:31.955526Z","shell.execute_reply.started":"2024-04-02T09:06:31.939073Z","shell.execute_reply":"2024-04-02T09:06:31.954752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Automatic Aggregation","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df): #Extract the maximum and minimum values for features P and A, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def num_expr_mean(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n\n        return expr_max\n    \n    \n    @staticmethod\n    def num_expr_std(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.std(col).alias(f\"std_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def date_expr(df): #Extract the maximum and minimum values for features D, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def str_expr(df): #Extract the maximum and minimum values for features M, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def other_expr(df): #Extract the maximum and minimum values for features T and L, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n    \n    @staticmethod\n    def count_expr(df): #Extract the maximum and minimum values for each num_group and add them as additional features.\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def get_exprs(df): #Execute the above function and return the result.\n        maxexprs = Aggregator.num_expr(df)[0] + \\\n                Aggregator.date_expr(df)[0] + \\\n                Aggregator.str_expr(df)[0] + \\\n                Aggregator.other_expr(df)[0] + \\\n                Aggregator.count_expr(df)[0]\n        \n        minexprs = Aggregator.num_expr(df)[1] + \\\n                Aggregator.date_expr(df)[1] + \\\n                Aggregator.str_expr(df)[1] + \\\n                Aggregator.other_expr(df)[1] + \\\n                Aggregator.count_expr(df)[1]\n\n        return maxexprs, minexprs","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:31.956759Z","iopub.execute_input":"2024-04-02T09:06:31.957026Z","iopub.status.idle":"2024-04-02T09:06:31.974916Z","shell.execute_reply.started":"2024-04-02T09:06:31.957005Z","shell.execute_reply":"2024-04-02T09:06:31.974127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File I/O","metadata":{}},{"cell_type":"code","source":"def read_file(path, depth=None): \n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        maxexprs, minexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs)\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        maxexprs, minexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:31.976017Z","iopub.execute_input":"2024-04-02T09:06:31.976285Z","iopub.status.idle":"2024-04-02T09:06:31.988439Z","shell.execute_reply.started":"2024-04-02T09:06:31.976264Z","shell.execute_reply":"2024-04-02T09:06:31.987589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:31.989406Z","iopub.execute_input":"2024-04-02T09:06:31.989671Z","iopub.status.idle":"2024-04-02T09:06:31.998272Z","shell.execute_reply.started":"2024-04-02T09:06:31.989641Z","shell.execute_reply":"2024-04-02T09:06:31.997427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:31.999497Z","iopub.execute_input":"2024-04-02T09:06:32.000267Z","iopub.status.idle":"2024-04-02T09:06:32.011953Z","shell.execute_reply.started":"2024-04-02T09:06:32.000236Z","shell.execute_reply":"2024-04-02T09:06:32.011161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:32.012913Z","iopub.execute_input":"2024-04-02T09:06:32.013152Z","iopub.status.idle":"2024-04-02T09:06:32.026121Z","shell.execute_reply.started":"2024-04-02T09:06:32.013131Z","shell.execute_reply":"2024-04-02T09:06:32.025243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\nDRY_RUN = True if sample.shape[0] == 10 else False   # if num of records of test data is 10, dry-run is enable.\nPRESETS = \"medium_quality\"\nMODEL_PATH = \"/kaggle/input/home-credit-automl-training/predictor\"","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:32.030171Z","iopub.execute_input":"2024-04-02T09:06:32.030420Z","iopub.status.idle":"2024-04-02T09:06:32.042676Z","shell.execute_reply.started":"2024-04-02T09:06:32.030399Z","shell.execute_reply":"2024-04-02T09:06:32.041923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:32.043919Z","iopub.execute_input":"2024-04-02T09:06:32.044426Z","iopub.status.idle":"2024-04-02T09:06:32.052346Z","shell.execute_reply.started":"2024-04-02T09:06:32.044396Z","shell.execute_reply":"2024-04-02T09:06:32.051566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:06:32.055214Z","iopub.execute_input":"2024-04-02T09:06:32.055584Z","iopub.status.idle":"2024-04-02T09:07:06.372380Z","shell.execute_reply.started":"2024-04-02T09:06:32.055559Z","shell.execute_reply":"2024-04-02T09:07:06.371353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:06.373585Z","iopub.execute_input":"2024-04-02T09:07:06.373891Z","iopub.status.idle":"2024-04-02T09:07:14.788631Z","shell.execute_reply.started":"2024-04-02T09:07:06.373866Z","shell.execute_reply":"2024-04-02T09:07:14.787717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:14.789949Z","iopub.execute_input":"2024-04-02T09:07:14.790345Z","iopub.status.idle":"2024-04-02T09:07:18.188099Z","shell.execute_reply.started":"2024-04-02T09:07:14.790306Z","shell.execute_reply":"2024-04-02T09:07:18.187186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\nprint(df_train.shape)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:18.189130Z","iopub.execute_input":"2024-04-02T09:07:18.189407Z","iopub.status.idle":"2024-04-02T09:07:39.532778Z","shell.execute_reply.started":"2024-04-02T09:07:18.189383Z","shell.execute_reply":"2024-04-02T09:07:39.531791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ndf_train = reduce_mem_usage(df_train)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:39.534048Z","iopub.execute_input":"2024-04-02T09:07:39.534400Z","iopub.status.idle":"2024-04-02T09:07:45.791010Z","shell.execute_reply.started":"2024-04-02T09:07:39.534367Z","shell.execute_reply":"2024-04-02T09:07:45.790068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"NUM=len(df_train)\n\nif DRY_RUN:\n    print(f\"df_train.shape : {df_train.shape} --> \", end=\"\")\n    df_train = df_train.iloc[:NUM]\n    print( df_train.shape )","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:45.792252Z","iopub.execute_input":"2024-04-02T09:07:45.792533Z","iopub.status.idle":"2024-04-02T09:07:45.799568Z","shell.execute_reply.started":"2024-04-02T09:07:45.792509Z","shell.execute_reply":"2024-04-02T09:07:45.798587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictor = TabularPredictor(\n#    label=\"target\",\n#    problem_type=\"binary\",\n#    eval_metric=\"roc_auc\",\n#    path=\"predictor\",\n#)\n\n# Load model\npredictor = TabularPredictor.load(path=MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:45.801014Z","iopub.execute_input":"2024-04-02T09:07:45.801365Z","iopub.status.idle":"2024-04-02T09:07:45.831841Z","shell.execute_reply.started":"2024-04-02T09:07:45.801341Z","shell.execute_reply":"2024-04-02T09:07:45.831160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#weeks = df_train[\"WEEK_NUM\"]\n#df_train = df_train.drop(columns=[\"case_id\", \"WEEK_NUM\"])\n#cv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n#for idx_train, idx_valid in cv.split(df_train, df_train[\"target\"], groups=weeks):\n#    fold_train = df_train.iloc[idx_train]\n#    fold_valid = df_train.iloc[idx_valid]\n#    train_data = TabularDataset(fold_train)\n#    valid_data = TabularDataset(fold_valid)\n#    break","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:45.832870Z","iopub.execute_input":"2024-04-02T09:07:45.833122Z","iopub.status.idle":"2024-04-02T09:07:45.837032Z","shell.execute_reply.started":"2024-04-02T09:07:45.833100Z","shell.execute_reply":"2024-04-02T09:07:45.836139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del df_train # It is deleted to reduce memory usage\n#gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:45.838324Z","iopub.execute_input":"2024-04-02T09:07:45.838675Z","iopub.status.idle":"2024-04-02T09:07:45.845605Z","shell.execute_reply.started":"2024-04-02T09:07:45.838645Z","shell.execute_reply":"2024-04-02T09:07:45.844928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%%time\n#predictor.fit(\n#    train_data,\n#    tuning_data=valid_data,\n#    save_space=True,\n#    presets=PRESETS,\n#    use_bag_holdout=True,\n#    ag_args_fit={'num_gpus': 1},\n#)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-02T09:07:45.846772Z","iopub.execute_input":"2024-04-02T09:07:45.847362Z","iopub.status.idle":"2024-04-02T09:07:45.855315Z","shell.execute_reply.started":"2024-04-02T09:07:45.847332Z","shell.execute_reply":"2024-04-02T09:07:45.854549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:45.856319Z","iopub.execute_input":"2024-04-02T09:07:45.856632Z","iopub.status.idle":"2024-04-02T09:07:45.864810Z","shell.execute_reply.started":"2024-04-02T09:07:45.856602Z","shell.execute_reply":"2024-04-02T09:07:45.864092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training result","metadata":{}},{"cell_type":"code","source":"predictor.leaderboard()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:45.865791Z","iopub.execute_input":"2024-04-02T09:07:45.866045Z","iopub.status.idle":"2024-04-02T09:07:45.888162Z","shell.execute_reply.started":"2024-04-02T09:07:45.866023Z","shell.execute_reply":"2024-04-02T09:07:45.887336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_lb = predictor.leaderboard()\nfrom matplotlib import pyplot as plt\nplt.scatter( df_lb[\"score_val\"], df_lb[\"model\"] )\nplt.grid()\nplt.xlabel(\"CV(roc_auc)\")\nplt.ylabel(\"Model name\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-02T09:07:45.889157Z","iopub.execute_input":"2024-04-02T09:07:45.889473Z","iopub.status.idle":"2024-04-02T09:07:46.143771Z","shell.execute_reply.started":"2024-04-02T09:07:45.889441Z","shell.execute_reply":"2024-04-02T09:07:46.142791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:46.145360Z","iopub.execute_input":"2024-04-02T09:07:46.145702Z","iopub.status.idle":"2024-04-02T09:07:46.243916Z","shell.execute_reply.started":"2024-04-02T09:07:46.145669Z","shell.execute_reply":"2024-04-02T09:07:46.242922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:46.245123Z","iopub.execute_input":"2024-04-02T09:07:46.245418Z","iopub.status.idle":"2024-04-02T09:07:46.283099Z","shell.execute_reply.started":"2024-04-02T09:07:46.245393Z","shell.execute_reply":"2024-04-02T09:07:46.282109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.select([col for col in df_test.columns if col != \"target\"])\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:46.284434Z","iopub.execute_input":"2024-04-02T09:07:46.284831Z","iopub.status.idle":"2024-04-02T09:07:46.292742Z","shell.execute_reply.started":"2024-04-02T09:07:46.284797Z","shell.execute_reply":"2024-04-02T09:07:46.291790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test, cat_cols = to_pandas(df_test, cat_cols) # cat_cols was created by train data","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:46.298530Z","iopub.execute_input":"2024-04-02T09:07:46.298935Z","iopub.status.idle":"2024-04-02T09:07:46.361616Z","shell.execute_reply.started":"2024-04-02T09:07:46.298912Z","shell.execute_reply":"2024-04-02T09:07:46.360761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_test.shape)\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:46.362829Z","iopub.execute_input":"2024-04-02T09:07:46.363209Z","iopub.status.idle":"2024-04-02T09:07:46.386497Z","shell.execute_reply.started":"2024-04-02T09:07:46.363174Z","shell.execute_reply":"2024-04-02T09:07:46.385630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ndf_test = reduce_mem_usage(df_test)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:46.387653Z","iopub.execute_input":"2024-04-02T09:07:46.387994Z","iopub.status.idle":"2024-04-02T09:07:46.799325Z","shell.execute_reply.started":"2024-04-02T09:07:46.387964Z","shell.execute_reply":"2024-04-02T09:07:46.798431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"case_id\", \"WEEK_NUM\"])\ntest_data = TabularDataset(df_test)\ny_pred = predictor.predict_proba(test_data).iloc[:, 1].values","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:46.800652Z","iopub.execute_input":"2024-04-02T09:07:46.801029Z","iopub.status.idle":"2024-04-02T09:07:48.460053Z","shell.execute_reply.started":"2024-04-02T09:07:46.800997Z","shell.execute_reply":"2024-04-02T09:07:48.459214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:48.461355Z","iopub.execute_input":"2024-04-02T09:07:48.461826Z","iopub.status.idle":"2024-04-02T09:07:48.469165Z","shell.execute_reply.started":"2024-04-02T09:07:48.461789Z","shell.execute_reply":"2024-04-02T09:07:48.468369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())\n\ndf_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:48.470191Z","iopub.execute_input":"2024-04-02T09:07:48.470477Z","iopub.status.idle":"2024-04-02T09:07:48.482140Z","shell.execute_reply.started":"2024-04-02T09:07:48.470455Z","shell.execute_reply":"2024-04-02T09:07:48.481259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-02T09:07:48.483123Z","iopub.execute_input":"2024-04-02T09:07:48.483403Z","iopub.status.idle":"2024-04-02T09:07:48.492404Z","shell.execute_reply.started":"2024-04-02T09:07:48.483369Z","shell.execute_reply":"2024-04-02T09:07:48.491486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}