{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":162314401,"sourceType":"kernelVersion"},{"sourceId":162317063,"sourceType":"kernelVersion"},{"sourceId":162351144,"sourceType":"kernelVersion"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Resource\n- training notebook\n  - https://www.kaggle.com/code/motono0223/home-credit-automl-training\n- packages for offline installation\n  - https://www.kaggle.com/code/motono0223/autogluon-pkgs\n  - https://www.kaggle.com/code/motono0223/ray-pkgs\n  \n# Reference \n- [1] [home-credit-baseline](https://www.kaggle.com/code/greysky/home-credit-baseline)\n- [2] [home-credit-baseline-max-min-features](https://www.kaggle.com/code/stechparme/home-credit-baseline-max-min-features)\n- [3] [dependency of autogluon (version confliction by ray package)](https://github.com/autogluon/autogluon/issues/3365)\n- [4] [Autogluon APIs](https://auto.gluon.ai/stable/api/autogluon.tabular.TabularPredictor.html)","metadata":{}},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/autogluon-pkgs autogluon > /dev/null","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-10T02:45:23.557162Z","iopub.execute_input":"2024-02-10T02:45:23.557504Z","iopub.status.idle":"2024-02-10T02:45:47.413693Z","shell.execute_reply.started":"2024-02-10T02:45:23.557452Z","shell.execute_reply":"2024-02-10T02:45:47.412518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/ray-pkgs --upgrade --force-reinstall -q ray==2.6.3","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-10T02:45:47.419786Z","iopub.execute_input":"2024-02-10T02:45:47.420055Z","iopub.status.idle":"2024-02-10T02:46:10.396589Z","shell.execute_reply.started":"2024-02-10T02:45:47.420028Z","shell.execute_reply":"2024-02-10T02:46:10.395571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom autogluon.tabular import TabularDataset, TabularPredictor","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:10.397908Z","iopub.execute_input":"2024-02-10T02:46:10.398209Z","iopub.status.idle":"2024-02-10T02:46:13.905044Z","shell.execute_reply.started":"2024-02-10T02:46:10.39818Z","shell.execute_reply":"2024-02-10T02:46:13.904277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df): #Standardize the dtype.\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df): #Change the feature for D to the difference in days from date_decision.\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df): #Remove those with an average is_null exceeding 0.95 and those that do not fall within the range 1 < nunique < 200.\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:13.907363Z","iopub.execute_input":"2024-02-10T02:46:13.909018Z","iopub.status.idle":"2024-02-10T02:46:14.045002Z","shell.execute_reply.started":"2024-02-10T02:46:13.908989Z","shell.execute_reply":"2024-02-10T02:46:14.043886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Automatic Aggregation","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df): #Extract the maximum and minimum values for features P and A, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def date_expr(df): #Extract the maximum and minimum values for features D, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def str_expr(df): #Extract the maximum and minimum values for features M, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def other_expr(df): #Extract the maximum and minimum values for features T and L, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n    \n    @staticmethod\n    def count_expr(df): #Extract the maximum and minimum values for each num_group and add them as additional features.\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        return expr_max, expr_min\n\n    @staticmethod\n    def get_exprs(df): #Execute the above function and return the result.\n        maxexprs = Aggregator.num_expr(df)[0] + \\\n                Aggregator.date_expr(df)[0] + \\\n                Aggregator.str_expr(df)[0] + \\\n                Aggregator.other_expr(df)[0] + \\\n                Aggregator.count_expr(df)[0]\n        \n        minexprs = Aggregator.num_expr(df)[1] + \\\n                Aggregator.date_expr(df)[1] + \\\n                Aggregator.str_expr(df)[1] + \\\n                Aggregator.other_expr(df)[1] + \\\n                Aggregator.count_expr(df)[1]\n\n        return maxexprs, minexprs","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:14.046322Z","iopub.execute_input":"2024-02-10T02:46:14.046702Z","iopub.status.idle":"2024-02-10T02:46:14.766971Z","shell.execute_reply.started":"2024-02-10T02:46:14.046666Z","shell.execute_reply":"2024-02-10T02:46:14.765865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File I/O","metadata":{}},{"cell_type":"code","source":"def read_file(path, depth=None): \n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        maxexprs, minexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs)\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        maxexprs, minexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:14.768565Z","iopub.execute_input":"2024-02-10T02:46:14.768903Z","iopub.status.idle":"2024-02-10T02:46:14.781528Z","shell.execute_reply.started":"2024-02-10T02:46:14.768874Z","shell.execute_reply":"2024-02-10T02:46:14.780733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:14.78256Z","iopub.execute_input":"2024-02-10T02:46:14.782843Z","iopub.status.idle":"2024-02-10T02:46:14.793199Z","shell.execute_reply.started":"2024-02-10T02:46:14.782802Z","shell.execute_reply":"2024-02-10T02:46:14.792361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:14.794271Z","iopub.execute_input":"2024-02-10T02:46:14.794592Z","iopub.status.idle":"2024-02-10T02:46:14.80266Z","shell.execute_reply.started":"2024-02-10T02:46:14.794566Z","shell.execute_reply":"2024-02-10T02:46:14.801816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:14.803938Z","iopub.execute_input":"2024-02-10T02:46:14.804302Z","iopub.status.idle":"2024-02-10T02:46:14.818011Z","shell.execute_reply.started":"2024-02-10T02:46:14.804269Z","shell.execute_reply":"2024-02-10T02:46:14.817168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\nDRY_RUN = True if sample.shape[0] == 10 else False   # if num of records of test data is 10, dry-run is enable.\nPRESETS = \"medium_quality\"\nMODEL_PATH = \"/kaggle/input/home-credit-automl-training/predictor\"","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:14.819111Z","iopub.execute_input":"2024-02-10T02:46:14.819939Z","iopub.status.idle":"2024-02-10T02:46:14.833864Z","shell.execute_reply.started":"2024-02-10T02:46:14.819913Z","shell.execute_reply":"2024-02-10T02:46:14.833129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:14.835388Z","iopub.execute_input":"2024-02-10T02:46:14.835677Z","iopub.status.idle":"2024-02-10T02:46:14.844265Z","shell.execute_reply.started":"2024-02-10T02:46:14.835653Z","shell.execute_reply":"2024-02-10T02:46:14.843523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:14.845257Z","iopub.execute_input":"2024-02-10T02:46:14.845539Z","iopub.status.idle":"2024-02-10T02:46:47.93426Z","shell.execute_reply.started":"2024-02-10T02:46:14.845509Z","shell.execute_reply":"2024-02-10T02:46:47.933428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:47.939456Z","iopub.execute_input":"2024-02-10T02:46:47.939761Z","iopub.status.idle":"2024-02-10T02:46:58.355676Z","shell.execute_reply.started":"2024-02-10T02:46:47.939735Z","shell.execute_reply":"2024-02-10T02:46:58.354675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:46:58.356854Z","iopub.execute_input":"2024-02-10T02:46:58.35716Z","iopub.status.idle":"2024-02-10T02:47:01.590123Z","shell.execute_reply.started":"2024-02-10T02:46:58.357131Z","shell.execute_reply":"2024-02-10T02:47:01.589219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\nprint(df_train.shape)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:47:01.59159Z","iopub.execute_input":"2024-02-10T02:47:01.591902Z","iopub.status.idle":"2024-02-10T02:47:21.662418Z","shell.execute_reply.started":"2024-02-10T02:47:01.591874Z","shell.execute_reply":"2024-02-10T02:47:21.661529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ndf_train = reduce_mem_usage(df_train)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:47:21.663768Z","iopub.execute_input":"2024-02-10T02:47:21.66415Z","iopub.status.idle":"2024-02-10T02:47:27.996411Z","shell.execute_reply.started":"2024-02-10T02:47:21.664114Z","shell.execute_reply":"2024-02-10T02:47:27.995506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"if DRY_RUN:\n    print(f\"df_train.shape : {df_train.shape} --> \", end=\"\")\n    df_train = df_train.iloc[:500]\n    print( df_train.shape )","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:47:27.997828Z","iopub.execute_input":"2024-02-10T02:47:27.998209Z","iopub.status.idle":"2024-02-10T02:47:28.006362Z","shell.execute_reply.started":"2024-02-10T02:47:27.998171Z","shell.execute_reply":"2024-02-10T02:47:28.00537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictor = TabularPredictor(\n#    label=\"target\",\n#    problem_type=\"binary\",\n#    eval_metric=\"roc_auc\",\n#    path=\"predictor\",\n#)\n\n# Load model\npredictor = TabularPredictor.load(path=MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:47:28.007618Z","iopub.execute_input":"2024-02-10T02:47:28.007928Z","iopub.status.idle":"2024-02-10T02:47:28.023848Z","shell.execute_reply.started":"2024-02-10T02:47:28.007902Z","shell.execute_reply":"2024-02-10T02:47:28.023034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#weeks = df_train[\"WEEK_NUM\"]\n#df_train = df_train.drop(columns=[\"case_id\", \"WEEK_NUM\"])\n#cv = StratifiedGroupKFold(n_splits=5, shuffle=False)\n#for idx_train, idx_valid in cv.split(df_train, df_train[\"target\"], groups=weeks):\n#    fold_train = df_train.iloc[idx_train]\n#    fold_valid = df_train.iloc[idx_valid]\n#    train_data = TabularDataset(fold_train)\n#    valid_data = TabularDataset(fold_valid)\n#    break","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:47:28.024983Z","iopub.execute_input":"2024-02-10T02:47:28.025246Z","iopub.status.idle":"2024-02-10T02:47:28.09791Z","shell.execute_reply.started":"2024-02-10T02:47:28.025222Z","shell.execute_reply":"2024-02-10T02:47:28.097018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del df_train # It is deleted to reduce memory usage\n#gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%%time\n#predictor.fit(\n#    train_data,\n#    tuning_data=valid_data,\n#    save_space=True,\n#    presets=PRESETS,\n#    use_bag_holdout=True,\n#    ag_args_fit={'num_gpus': 1},\n#)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-10T02:47:28.099022Z","iopub.execute_input":"2024-02-10T02:47:28.099294Z","iopub.status.idle":"2024-02-10T02:48:27.86199Z","shell.execute_reply.started":"2024-02-10T02:47:28.09927Z","shell.execute_reply":"2024-02-10T02:48:27.861267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:27.863162Z","iopub.execute_input":"2024-02-10T02:48:27.86381Z","iopub.status.idle":"2024-02-10T02:48:28.078208Z","shell.execute_reply.started":"2024-02-10T02:48:27.863781Z","shell.execute_reply":"2024-02-10T02:48:28.07722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training result","metadata":{}},{"cell_type":"code","source":"predictor.leaderboard()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:28.079623Z","iopub.execute_input":"2024-02-10T02:48:28.079959Z","iopub.status.idle":"2024-02-10T02:48:28.108593Z","shell.execute_reply.started":"2024-02-10T02:48:28.07993Z","shell.execute_reply":"2024-02-10T02:48:28.107534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_lb = predictor.leaderboard()\nfrom matplotlib import pyplot as plt\nplt.scatter( df_lb[\"score_val\"], df_lb[\"model\"] )\nplt.grid()\nplt.xlabel(\"CV(roc_auc)\")\nplt.ylabel(\"Model name\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:51:23.265621Z","iopub.execute_input":"2024-02-10T02:51:23.266019Z","iopub.status.idle":"2024-02-10T02:51:23.546485Z","shell.execute_reply.started":"2024-02-10T02:51:23.26598Z","shell.execute_reply":"2024-02-10T02:51:23.545525Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:28.110424Z","iopub.execute_input":"2024-02-10T02:48:28.110874Z","iopub.status.idle":"2024-02-10T02:48:28.337064Z","shell.execute_reply.started":"2024-02-10T02:48:28.11083Z","shell.execute_reply":"2024-02-10T02:48:28.336186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:28.338276Z","iopub.execute_input":"2024-02-10T02:48:28.338615Z","iopub.status.idle":"2024-02-10T02:48:28.377767Z","shell.execute_reply.started":"2024-02-10T02:48:28.338586Z","shell.execute_reply":"2024-02-10T02:48:28.376924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.select([col for col in df_test.columns if col != \"target\"])\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:28.37902Z","iopub.execute_input":"2024-02-10T02:48:28.379298Z","iopub.status.idle":"2024-02-10T02:48:28.386911Z","shell.execute_reply.started":"2024-02-10T02:48:28.379272Z","shell.execute_reply":"2024-02-10T02:48:28.386002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test, cat_cols = to_pandas(df_test, cat_cols) # cat_cols was created by train data","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:28.388262Z","iopub.execute_input":"2024-02-10T02:48:28.388811Z","iopub.status.idle":"2024-02-10T02:48:28.455392Z","shell.execute_reply.started":"2024-02-10T02:48:28.388783Z","shell.execute_reply":"2024-02-10T02:48:28.454505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_test.shape)\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:28.456707Z","iopub.execute_input":"2024-02-10T02:48:28.457875Z","iopub.status.idle":"2024-02-10T02:48:28.482038Z","shell.execute_reply.started":"2024-02-10T02:48:28.457846Z","shell.execute_reply":"2024-02-10T02:48:28.480922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ndf_test = reduce_mem_usage(df_test)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:28.483324Z","iopub.execute_input":"2024-02-10T02:48:28.483647Z","iopub.status.idle":"2024-02-10T02:48:28.943783Z","shell.execute_reply.started":"2024-02-10T02:48:28.48361Z","shell.execute_reply":"2024-02-10T02:48:28.942867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"case_id\", \"WEEK_NUM\"])\ntest_data = TabularDataset(df_test)\ny_pred = predictor.predict_proba(test_data).iloc[:, 1].values","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:28.94502Z","iopub.execute_input":"2024-02-10T02:48:28.945314Z","iopub.status.idle":"2024-02-10T02:48:29.071279Z","shell.execute_reply.started":"2024-02-10T02:48:28.945288Z","shell.execute_reply":"2024-02-10T02:48:29.070222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:29.072777Z","iopub.execute_input":"2024-02-10T02:48:29.073122Z","iopub.status.idle":"2024-02-10T02:48:29.080065Z","shell.execute_reply.started":"2024-02-10T02:48:29.073093Z","shell.execute_reply":"2024-02-10T02:48:29.079266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())\n\ndf_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:29.083243Z","iopub.execute_input":"2024-02-10T02:48:29.083682Z","iopub.status.idle":"2024-02-10T02:48:29.093817Z","shell.execute_reply.started":"2024-02-10T02:48:29.083654Z","shell.execute_reply":"2024-02-10T02:48:29.092877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-10T02:48:29.095227Z","iopub.execute_input":"2024-02-10T02:48:29.095588Z","iopub.status.idle":"2024-02-10T02:48:29.105693Z","shell.execute_reply.started":"2024-02-10T02:48:29.095559Z","shell.execute_reply":"2024-02-10T02:48:29.104736Z"},"trusted":true},"execution_count":null,"outputs":[]}]}