{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":7629125,"sourceType":"datasetVersion","datasetId":4424773},{"sourceId":7630675,"sourceType":"datasetVersion","datasetId":4446060},{"sourceId":162314401,"sourceType":"kernelVersion"},{"sourceId":162317063,"sourceType":"kernelVersion"},{"sourceId":162351144,"sourceType":"kernelVersion"}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Adding more features for AutoML models\n  \n<div class=\"alert alert-block alert-warning\" style=\"font-size:14px; font-family:verdana; line-height: 1.7em;\">\n    📌 &nbsp; My idea here was to add more aggregate features besides min and max: sum and var were also used.\n</div>\n\n<div class=\"alert alert-block alert-warning\" style=\"font-size:14px; font-family:verdana; line-height: 1.7em;\">\n    📌 &nbsp; Second Idea: save big df to csv, reset enviroment, read csv in chunks, then predict on each chunk \n</div>\n# Resource\n- [Training notebook](https://www.kaggle.com/code/vladislavkolesov/home-credit-automl-training)\n  \n# Reference \n- [1] [home-credit-baseline](https://www.kaggle.com/code/greysky/home-credit-baseline)\n- [2] [home-credit-baseline-max-min-features](https://www.kaggle.com/code/stechparme/home-credit-baseline-max-min-features)\n- [3] [dependency of autogluon (version confliction by ray package)](https://github.com/autogluon/autogluon/issues/3365)\n- [4] [Autogluon APIs](https://auto.gluon.ai/stable/api/autogluon.tabular.TabularPredictor.html)\n- [5] reference training notebook\n  - https://www.kaggle.com/code/motono0223/home-credit-automl-training\n- [6] packages for offline installation\n  - https://www.kaggle.com/code/motono0223/autogluon-pkgs\n  - https://www.kaggle.com/code/motono0223/ray-pkgs","metadata":{}},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/autogluon-pkgs autogluon > /dev/null\n!python -m pip install --no-index --find-links=/kaggle/input/ray-pkgs --upgrade --force-reinstall -q ray==2.6.3","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-19T17:24:40.349653Z","iopub.execute_input":"2024-02-19T17:24:40.350349Z","iopub.status.idle":"2024-02-19T17:28:29.155790Z","shell.execute_reply.started":"2024-02-19T17:24:40.350317Z","shell.execute_reply":"2024-02-19T17:28:29.154826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom autogluon.tabular import TabularDataset, TabularPredictor","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:29.157935Z","iopub.execute_input":"2024-02-19T17:28:29.158257Z","iopub.status.idle":"2024-02-19T17:28:34.037057Z","shell.execute_reply.started":"2024-02-19T17:28:29.158227Z","shell.execute_reply":"2024-02-19T17:28:34.036264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df): #Standardize the dtype.\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df): #Change the feature for D to the difference in days from date_decision.\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df): #Remove those with an average is_null exceeding 0.95 and those that do not fall within the range 1 < nunique < 200.\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.038139Z","iopub.execute_input":"2024-02-19T17:28:34.038743Z","iopub.status.idle":"2024-02-19T17:28:34.051090Z","shell.execute_reply.started":"2024-02-19T17:28:34.038716Z","shell.execute_reply":"2024-02-19T17:28:34.050098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Automatic Aggregation\nThat's where I added var and sum features. Note that sin, cos, mean also may imporove quality but just adding too many features might result in out of memory problems.","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df): #Extract the maximum and minimum values for features P and A, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n\n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def date_expr(df): #Extract the maximum and minimum values for features D, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        \n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def str_expr(df): #Extract the maximum and minimum values for features M, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        \n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def other_expr(df): #Extract the maximum and minimum values for features T and L, and add them as additional features.\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        \n        return expr_max, expr_min, expr_var, expr_sum\n    \n    @staticmethod\n    def count_expr(df): #Extract the maximum and minimum values for each num_group and add them as additional features.\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        \n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def get_exprs(df): #Execute the above function and return the result.\n        maxexprs = Aggregator.num_expr(df)[0] + \\\n                Aggregator.date_expr(df)[0] + \\\n                Aggregator.str_expr(df)[0] + \\\n                Aggregator.other_expr(df)[0] + \\\n                Aggregator.count_expr(df)[0]\n        \n        minexprs = Aggregator.num_expr(df)[1] + \\\n                Aggregator.date_expr(df)[1] + \\\n                Aggregator.str_expr(df)[1] + \\\n                Aggregator.other_expr(df)[1] + \\\n                Aggregator.count_expr(df)[1]\n        \n        varexprs = Aggregator.num_expr(df)[2] + \\\n                Aggregator.date_expr(df)[2] + \\\n                Aggregator.str_expr(df)[2] + \\\n                Aggregator.other_expr(df)[2] + \\\n                Aggregator.count_expr(df)[2]\n        \n        sumexprs = Aggregator.num_expr(df)[3] + \\\n                Aggregator.date_expr(df)[3] + \\\n                Aggregator.str_expr(df)[3] + \\\n                Aggregator.other_expr(df)[3] + \\\n                Aggregator.count_expr(df)[3]\n\n        return maxexprs, minexprs, varexprs, sumexprs","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.054048Z","iopub.execute_input":"2024-02-19T17:28:34.054375Z","iopub.status.idle":"2024-02-19T17:28:34.100664Z","shell.execute_reply.started":"2024-02-19T17:28:34.054345Z","shell.execute_reply":"2024-02-19T17:28:34.099814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File I/O","metadata":{}},{"cell_type":"code","source":"def read_file(path, depth=None): \n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        maxexprs, minexprs, varexpres, sumexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs, *varexpres, *sumexprs)\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        maxexprs, minexprs, varexpres, sumexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs, *varexpres, *sumexprs)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.101828Z","iopub.execute_input":"2024-02-19T17:28:34.102153Z","iopub.status.idle":"2024-02-19T17:28:34.113752Z","shell.execute_reply.started":"2024-02-19T17:28:34.102128Z","shell.execute_reply":"2024-02-19T17:28:34.112945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n\n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.114838Z","iopub.execute_input":"2024-02-19T17:28:34.115178Z","iopub.status.idle":"2024-02-19T17:28:34.125649Z","shell.execute_reply.started":"2024-02-19T17:28:34.115154Z","shell.execute_reply":"2024-02-19T17:28:34.124897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n\n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.126744Z","iopub.execute_input":"2024-02-19T17:28:34.127329Z","iopub.status.idle":"2024-02-19T17:28:34.134524Z","shell.execute_reply.started":"2024-02-19T17:28:34.127297Z","shell.execute_reply":"2024-02-19T17:28:34.133811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    gc.collect()\n    # you shouldn't return df because source df gets changed and never deletes from RAM\n    # return df","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.135586Z","iopub.execute_input":"2024-02-19T17:28:34.135870Z","iopub.status.idle":"2024-02-19T17:28:34.149625Z","shell.execute_reply.started":"2024-02-19T17:28:34.135846Z","shell.execute_reply":"2024-02-19T17:28:34.148732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\nDRY_RUN = True if sample.shape[0] == 10 else False   # if num of records of test data is 10, dry-run is enable.\nPRESETS = \"medium_quality\"\nMODEL_PATH = \"/kaggle/input/home-credit-automl-models-dataset/\"","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.150783Z","iopub.execute_input":"2024-02-19T17:28:34.151362Z","iopub.status.idle":"2024-02-19T17:28:34.169257Z","shell.execute_reply.started":"2024-02-19T17:28:34.151306Z","shell.execute_reply":"2024-02-19T17:28:34.168478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.172889Z","iopub.execute_input":"2024-02-19T17:28:34.173182Z","iopub.status.idle":"2024-02-19T17:28:34.177429Z","shell.execute_reply.started":"2024-02-19T17:28:34.173157Z","shell.execute_reply":"2024-02-19T17:28:34.176395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"predictor = TabularPredictor.load(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.178892Z","iopub.execute_input":"2024-02-19T17:28:34.179421Z","iopub.status.idle":"2024-02-19T17:28:34.793456Z","shell.execute_reply.started":"2024-02-19T17:28:34.179362Z","shell.execute_reply":"2024-02-19T17:28:34.792666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training result","metadata":{}},{"cell_type":"code","source":"predictor.leaderboard()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:34.794504Z","iopub.execute_input":"2024-02-19T17:28:34.795089Z","iopub.status.idle":"2024-02-19T17:28:34.828259Z","shell.execute_reply.started":"2024-02-19T17:28:34.795054Z","shell.execute_reply":"2024-02-19T17:28:34.827378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_lb = predictor.leaderboard()\nfrom matplotlib import pyplot as plt\nplt.scatter( df_lb[\"score_val\"], df_lb[\"model\"] )\nplt.grid()\nplt.xlabel(\"CV(roc_auc)\")\nplt.ylabel(\"Model name\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-19T17:28:34.829488Z","iopub.execute_input":"2024-02-19T17:28:34.829863Z","iopub.status.idle":"2024-02-19T17:28:35.043213Z","shell.execute_reply.started":"2024-02-19T17:28:34.829837Z","shell.execute_reply":"2024-02-19T17:28:35.042323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_lb","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:35.044568Z","iopub.execute_input":"2024-02-19T17:28:35.044827Z","iopub.status.idle":"2024-02-19T17:28:35.049124Z","shell.execute_reply.started":"2024-02-19T17:28:35.044805Z","shell.execute_reply":"2024-02-19T17:28:35.048149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Files Read & Feature Engineering","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:35.050432Z","iopub.execute_input":"2024-02-19T17:28:35.050978Z","iopub.status.idle":"2024-02-19T17:28:35.837506Z","shell.execute_reply.started":"2024-02-19T17:28:35.050946Z","shell.execute_reply":"2024-02-19T17:28:35.836495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store).drop(columns=\"target\")\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:35.838740Z","iopub.execute_input":"2024-02-19T17:28:35.839093Z","iopub.status.idle":"2024-02-19T17:28:47.115738Z","shell.execute_reply.started":"2024-02-19T17:28:35.839058Z","shell.execute_reply":"2024-02-19T17:28:47.114887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nwith open('/kaggle/input/home-credit-cat-cols-from-train-df/cat_cols.pickle', 'rb') as f:\n    cat_cols = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:47.116828Z","iopub.execute_input":"2024-02-19T17:28:47.117090Z","iopub.status.idle":"2024-02-19T17:28:47.127008Z","shell.execute_reply.started":"2024-02-19T17:28:47.117067Z","shell.execute_reply":"2024-02-19T17:28:47.126276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test, cat_cols = to_pandas(df_test, cat_cols=cat_cols) # cat_cols was created by train data","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:28:47.128201Z","iopub.execute_input":"2024-02-19T17:28:47.128619Z","iopub.status.idle":"2024-02-19T17:29:04.283214Z","shell.execute_reply.started":"2024-02-19T17:28:47.128588Z","shell.execute_reply":"2024-02-19T17:29:04.282353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\n#reduce_mem_usage(df_test)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:29:04.284405Z","iopub.execute_input":"2024-02-19T17:29:04.284707Z","iopub.status.idle":"2024-02-19T17:29:04.434461Z","shell.execute_reply.started":"2024-02-19T17:29:04.284683Z","shell.execute_reply":"2024-02-19T17:29:04.432892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We have no so much space left (I changed test_base to train base to imitate hidden test size) \n  \n#### From Data description: \n- Test Files:\n\n    - test_base.csv (**Note, the hidden test_base.csv contains approximately 90% of the numbers of case_id values of train_base.csv**)","metadata":{}},{"cell_type":"markdown","source":"1. ### Save big df to .csv file:","metadata":{}},{"cell_type":"code","source":"test_data = TabularDataset(df_test.drop(columns=[\"case_id\", \"WEEK_NUM\"]))\ndel df_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:29:04.435646Z","iopub.execute_input":"2024-02-19T17:29:04.435952Z","iopub.status.idle":"2024-02-19T17:29:11.278883Z","shell.execute_reply.started":"2024-02-19T17:29:04.435926Z","shell.execute_reply":"2024-02-19T17:29:11.278022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.to_csv('test_data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:29:11.280082Z","iopub.execute_input":"2024-02-19T17:29:11.280373Z","iopub.status.idle":"2024-02-19T17:38:43.387744Z","shell.execute_reply.started":"2024-02-19T17:29:11.280349Z","shell.execute_reply":"2024-02-19T17:38:43.386442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2. ### Reset enviroment to free RAM:\n- Note that all variables will be deleted, also you'll have to import all packages again","metadata":{}},{"cell_type":"code","source":"%reset -f\n!mkdir -p submission_data\n!mv test_data.csv submission_data/test_data.csv  # move file to submission_data dir\nimport glob\nfor f in glob.glob('*'):\n    if not f.startswith('submission_data'):# delete all files if path not starts with submission_data\n        !rm -rf {f}","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:38:43.388865Z","iopub.execute_input":"2024-02-19T17:38:43.389151Z","iopub.status.idle":"2024-02-19T17:38:46.467575Z","shell.execute_reply.started":"2024-02-19T17:38:43.389127Z","shell.execute_reply":"2024-02-19T17:38:46.466518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3. ### Read+predict in chunks, [reference](https://auto.gluon.ai/0.7.0/tutorials/tabular_prediction/tabular-faq.html#how-can-i-perform-inference-on-a-file-that-wont-fit-in-memory)","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom autogluon.tabular import TabularDataset, TabularPredictor","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:38:46.469335Z","iopub.execute_input":"2024-02-19T17:38:46.470298Z","iopub.status.idle":"2024-02-19T17:38:46.477063Z","shell.execute_reply.started":"2024-02-19T17:38:46.470258Z","shell.execute_reply":"2024-02-19T17:38:46.476248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PATH = \"/kaggle/input/home-credit-automl-models-dataset/\"\npredictor = TabularPredictor.load(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:38:46.478262Z","iopub.execute_input":"2024-02-19T17:38:46.478605Z","iopub.status.idle":"2024-02-19T17:38:46.520840Z","shell.execute_reply.started":"2024-02-19T17:38:46.478573Z","shell.execute_reply":"2024-02-19T17:38:46.520016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CHUNK_SIZE = 10 ** 6\nreader = pd.read_csv('/kaggle/working/submission_data/test_data.csv', chunksize=CHUNK_SIZE)\ny_pred = []\nfor df_chunk in reader:\n    y_pred.append(predictor.predict_proba(df_chunk).iloc[:, 1].values)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:38:46.521975Z","iopub.execute_input":"2024-02-19T17:38:46.522309Z","iopub.status.idle":"2024-02-19T17:49:52.803422Z","shell.execute_reply.started":"2024-02-19T17:38:46.522279Z","shell.execute_reply":"2024-02-19T17:49:52.801766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = np.concatenate(y_pred, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:53:22.962187Z","iopub.execute_input":"2024-02-19T17:53:22.962931Z","iopub.status.idle":"2024-02-19T17:53:22.970234Z","shell.execute_reply.started":"2024-02-19T17:53:22.962899Z","shell.execute_reply":"2024-02-19T17:53:22.969296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"del reader\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:53:45.813700Z","iopub.execute_input":"2024-02-19T17:53:45.814522Z","iopub.status.idle":"2024-02-19T17:53:46.027296Z","shell.execute_reply.started":"2024-02-19T17:53:45.814489Z","shell.execute_reply":"2024-02-19T17:53:46.026350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\nprint(\"Check null: \", (y_pred == None).any())","metadata":{"execution":{"iopub.status.busy":"2024-02-19T17:55:47.452795Z","iopub.execute_input":"2024-02-19T17:55:47.453203Z","iopub.status.idle":"2024-02-19T17:55:47.597121Z","shell.execute_reply.started":"2024-02-19T17:55:47.453174Z","shell.execute_reply":"2024-02-19T17:55:47.596067Z"},"trusted":true},"execution_count":null,"outputs":[]}]}