{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7630675,"sourceType":"datasetVersion","datasetId":4446060},{"sourceId":7629125,"sourceType":"datasetVersion","datasetId":4424773,"isSourceIdPinned":true},{"sourceId":162314401,"sourceType":"kernelVersion"},{"sourceId":162317063,"sourceType":"kernelVersion"},{"sourceId":162351144,"sourceType":"kernelVersion"}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-success\" align=\"center\" style=\"font-size:16px; font-family:verdana; line-height: 1.7em;\">\n Adding more features for AutoML models\n</div>\n<div class=\"alert alert-block alert-danger\" align=\"center\" style=\"font-size:16px; font-family:verdana; line-height: 1.7em;\">\n❤️ Dont forget to ▲upvote▲ if you find this notebook usefull! ❤️ \n</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-warning\" style=\"font-size:14px; font-family:verdana; line-height: 1.7em;\">\n    📌 &nbsp; My idea here was to add more aggregate features besides min and max: sum and var were also used.\n</div>\n\n<div class=\"alert alert-block alert-warning\" style=\"font-size:14px; font-family:verdana; line-height: 1.7em;\">\n    📌 &nbsp; Second Idea: save big df to csv, reset enviroment, read csv in chunks, then predict on each chunk \n</div>\n\n# Resources\n- ### [Training notebook](https://www.kaggle.com/code/vladislavkolesov/home-credit-automl-training)\n- ### more usefull tricks to work with big dataframes efficiently, [discussion](https://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/discussion/475485)\n\n# Reference \n- ### [1] [home-credit-baseline](https://www.kaggle.com/code/greysky/home-credit-baseline)\n- ### [2] [home-credit-baseline-max-min-features](https://www.kaggle.com/code/stechparme/home-credit-baseline-max-min-features)\n- ### [3] [dependency of autogluon (version confliction by ray package)](https://github.com/autogluon/autogluon/issues/3365)\n- ### [4] [Autogluon APIs](https://auto.gluon.ai/stable/api/autogluon.tabular.TabularPredictor.html)\n- ### [5] reference training notebook\n  - https://www.kaggle.com/code/motono0223/home-credit-automl-training\n- ### [6] packages for offline installation\n  - https://www.kaggle.com/code/motono0223/autogluon-pkgs\n  - https://www.kaggle.com/code/motono0223/ray-pkgs\n- ### [7] polars [documentation](https://docs.pola.rs/py-polars/html/reference/) - focuses on utilizing hardware and *parallel processing capabilities* to offer high performance on a single machine","metadata":{}},{"cell_type":"markdown","source":"## Install and import libraries <a name=\"install_and_import\"></a>","metadata":{}},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/autogluon-pkgs autogluon > /dev/null\n!python -m pip install --no-index --find-links=/kaggle/input/ray-pkgs --upgrade --force-reinstall -q ray==2.6.3","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-03-12T18:13:25.004959Z","iopub.execute_input":"2024-03-12T18:13:25.005256Z","iopub.status.idle":"2024-03-12T18:17:19.070750Z","shell.execute_reply.started":"2024-03-12T18:13:25.005232Z","shell.execute_reply":"2024-03-12T18:17:19.069594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom autogluon.tabular import TabularDataset, TabularPredictor","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:19.072817Z","iopub.execute_input":"2024-03-12T18:17:19.073335Z","iopub.status.idle":"2024-03-12T18:17:23.983020Z","shell.execute_reply.started":"2024-03-12T18:17:19.073271Z","shell.execute_reply":"2024-03-12T18:17:23.982151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline <a name=\"pipeline\"></a>","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df): #Standardize the dtype.\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df): #Change the feature for D to the difference in days from date_decision.\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df): #Remove those with an average is_null exceeding 0.95 and those that do not fall within the range 1 < nunique < 200.\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:23.984164Z","iopub.execute_input":"2024-03-12T18:17:23.984794Z","iopub.status.idle":"2024-03-12T18:17:23.997720Z","shell.execute_reply.started":"2024-03-12T18:17:23.984767Z","shell.execute_reply":"2024-03-12T18:17:23.996957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Automatic Aggregation and new features <a name=\"auto_agg_and_new_feats\"></a> \nThat's where I added var and sum features. Note that sin, cos, mean also may imporove quality but just adding too many features might result in out of memory problems.","metadata":{}},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df): \n        \"\"\"Extract the maximum, minimum, variance and sum for features P and A,\n            and add them as additional features.\n        \"\"\"\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n\n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def date_expr(df): \n        \"\"\"Extract the maximum and minimum, variance and sum for features D,\n            and add them as additional features.\n        \"\"\"\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        \n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def str_expr(df): \n        \"\"\"Extract the maximum and minimum, variance and sum for features D,\n            and add them as additional features.\n        \"\"\"\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        \n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def other_expr(df): \n        \"\"\"Extract the maximum and minimum, variance and sum for features T, L\n            and add them as additional features.\n        \"\"\"\n        \n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        \n        return expr_max, expr_min, expr_var, expr_sum\n    \n    @staticmethod\n    def count_expr(df):\n        \"\"\"Extract the maximum and minimum, variance and sum for each num_group\n            and add them as additional features.\n        \"\"\"\n        \n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n        expr_sum = [pl.sum(col).alias(f\"sum_{col}\") for col in cols]\n        \n        return expr_max, expr_min, expr_var, expr_sum\n\n    @staticmethod\n    def get_exprs(df): \n        \"\"\"Execute the above function and return the result.\n        \"\"\"\n        \n        maxexprs = Aggregator.num_expr(df)[0] + \\\n                Aggregator.date_expr(df)[0] + \\\n                Aggregator.str_expr(df)[0] + \\\n                Aggregator.other_expr(df)[0] + \\\n                Aggregator.count_expr(df)[0]\n        \n        minexprs = Aggregator.num_expr(df)[1] + \\\n                Aggregator.date_expr(df)[1] + \\\n                Aggregator.str_expr(df)[1] + \\\n                Aggregator.other_expr(df)[1] + \\\n                Aggregator.count_expr(df)[1]\n        \n        varexprs = Aggregator.num_expr(df)[2] + \\\n                Aggregator.date_expr(df)[2] + \\\n                Aggregator.str_expr(df)[2] + \\\n                Aggregator.other_expr(df)[2] + \\\n                Aggregator.count_expr(df)[2]\n        \n        sumexprs = Aggregator.num_expr(df)[3] + \\\n                Aggregator.date_expr(df)[3] + \\\n                Aggregator.str_expr(df)[3] + \\\n                Aggregator.other_expr(df)[3] + \\\n                Aggregator.count_expr(df)[3]\n\n        return maxexprs, minexprs, varexprs, sumexprs","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.000565Z","iopub.execute_input":"2024-03-12T18:17:24.001172Z","iopub.status.idle":"2024-03-12T18:17:24.063473Z","shell.execute_reply.started":"2024-03-12T18:17:24.001132Z","shell.execute_reply":"2024-03-12T18:17:24.062676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File I/O <a name=\"file_io\"></a> ","metadata":{}},{"cell_type":"code","source":"def read_file(path, depth=None): \n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        maxexprs, minexprs, varexpres, sumexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs, *varexpres, *sumexprs)\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        maxexprs, minexprs, varexpres, sumexprs = Aggregator.get_exprs(df)\n        df = df.group_by(\"case_id\").agg(*maxexprs, *minexprs, *varexpres, *sumexprs)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.064594Z","iopub.execute_input":"2024-03-12T18:17:24.065217Z","iopub.status.idle":"2024-03-12T18:17:24.078797Z","shell.execute_reply.started":"2024-03-12T18:17:24.065185Z","shell.execute_reply":"2024-03-12T18:17:24.078108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n\n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.079797Z","iopub.execute_input":"2024-03-12T18:17:24.080070Z","iopub.status.idle":"2024-03-12T18:17:24.091859Z","shell.execute_reply.started":"2024-03-12T18:17:24.080048Z","shell.execute_reply":"2024-03-12T18:17:24.090949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n\n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.092948Z","iopub.execute_input":"2024-03-12T18:17:24.093347Z","iopub.status.idle":"2024-03-12T18:17:24.101114Z","shell.execute_reply.started":"2024-03-12T18:17:24.093317Z","shell.execute_reply":"2024-03-12T18:17:24.100241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" Iterate through all the columns of a dataframe and modify the data type\n        to reduce memory (disk) usage\n        \n        May increace RAM usage! Use with care.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    gc.collect()\n    # you don't have to return df\n    # return df","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.102088Z","iopub.execute_input":"2024-03-12T18:17:24.102321Z","iopub.status.idle":"2024-03-12T18:17:24.115636Z","shell.execute_reply.started":"2024-03-12T18:17:24.102300Z","shell.execute_reply":"2024-03-12T18:17:24.114735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\nDRY_RUN = True if sample.shape[0] == 10 else False   # if num of records of test data is 10, dry-run is enable.\nPRESETS = \"medium_quality\"\nMODEL_PATH = \"/kaggle/input/home-credit-automl-models-dataset/\"","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.117037Z","iopub.execute_input":"2024-03-12T18:17:24.117522Z","iopub.status.idle":"2024-03-12T18:17:24.134856Z","shell.execute_reply.started":"2024-03-12T18:17:24.117493Z","shell.execute_reply":"2024-03-12T18:17:24.134106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.139582Z","iopub.execute_input":"2024-03-12T18:17:24.139882Z","iopub.status.idle":"2024-03-12T18:17:24.143982Z","shell.execute_reply.started":"2024-03-12T18:17:24.139860Z","shell.execute_reply":"2024-03-12T18:17:24.143187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"predictor = TabularPredictor.load(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.144957Z","iopub.execute_input":"2024-03-12T18:17:24.145203Z","iopub.status.idle":"2024-03-12T18:17:24.750143Z","shell.execute_reply.started":"2024-03-12T18:17:24.145181Z","shell.execute_reply":"2024-03-12T18:17:24.749353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training result","metadata":{}},{"cell_type":"code","source":"predictor.leaderboard()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:24.751205Z","iopub.execute_input":"2024-03-12T18:17:24.751831Z","iopub.status.idle":"2024-03-12T18:17:24.787990Z","shell.execute_reply.started":"2024-03-12T18:17:24.751803Z","shell.execute_reply":"2024-03-12T18:17:24.787121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_lb = predictor.leaderboard()\nfrom matplotlib import pyplot as plt\nplt.scatter( df_lb[\"score_val\"], df_lb[\"model\"] )\nplt.grid()\nplt.xlabel(\"CV(roc_auc)\")\nplt.ylabel(\"Model name\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-12T18:17:24.788972Z","iopub.execute_input":"2024-03-12T18:17:24.789208Z","iopub.status.idle":"2024-03-12T18:17:25.005138Z","shell.execute_reply.started":"2024-03-12T18:17:24.789187Z","shell.execute_reply":"2024-03-12T18:17:25.004250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_lb","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.006266Z","iopub.execute_input":"2024-03-12T18:17:25.006553Z","iopub.status.idle":"2024-03-12T18:17:25.010611Z","shell.execute_reply.started":"2024-03-12T18:17:25.006528Z","shell.execute_reply":"2024-03-12T18:17:25.009705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Files Read & Feature Engineering\nNote: not all files used here so far, for example, test + train_person_2.csv\n","metadata":{}},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.011692Z","iopub.execute_input":"2024-03-12T18:17:25.012028Z","iopub.status.idle":"2024-03-12T18:17:25.310242Z","shell.execute_reply.started":"2024-03-12T18:17:25.011999Z","shell.execute_reply":"2024-03-12T18:17:25.309303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store).drop(columns=\"target\")\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.311453Z","iopub.execute_input":"2024-03-12T18:17:25.311807Z","iopub.status.idle":"2024-03-12T18:17:25.394392Z","shell.execute_reply.started":"2024-03-12T18:17:25.311774Z","shell.execute_reply":"2024-03-12T18:17:25.393550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nwith open('/kaggle/input/home-credit-cat-cols-from-train-df/cat_cols.pickle', 'rb') as f:\n    cat_cols = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.395550Z","iopub.execute_input":"2024-03-12T18:17:25.395843Z","iopub.status.idle":"2024-03-12T18:17:25.408133Z","shell.execute_reply.started":"2024-03-12T18:17:25.395819Z","shell.execute_reply":"2024-03-12T18:17:25.407425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test, cat_cols = to_pandas(df_test, cat_cols=cat_cols) # cat_cols was created by train data","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.409092Z","iopub.execute_input":"2024-03-12T18:17:25.409353Z","iopub.status.idle":"2024-03-12T18:17:25.486856Z","shell.execute_reply.started":"2024-03-12T18:17:25.409331Z","shell.execute_reply":"2024-03-12T18:17:25.486151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\n#reduce_mem_usage(df_test)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.487913Z","iopub.execute_input":"2024-03-12T18:17:25.488231Z","iopub.status.idle":"2024-03-12T18:17:25.622623Z","shell.execute_reply.started":"2024-03-12T18:17:25.488201Z","shell.execute_reply":"2024-03-12T18:17:25.621686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We have no so much space left (you can change test_base to train base to imitate hidden test size or check [here](https://www.kaggle.com/code/vladislavkolesov/read-data-in-ckunks-to-save-ram-home-credit)) \n  \n#### From Data description: \n- Test Files:\n\n    - test_base.csv (**Note, the hidden test_base.csv contains approximately *90%* of the numbers of case_id values of train_base.csv**)\n    \n### But I decided to imitate whole train size","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-warning\" style=\"font-size:14px; font-family:verdana; line-height: 1.7em;\"\n## Steps:\n- #### [Save big df to .csv file](#step1)\n- #### [Reset enviroment to free RAM](#step2)\n- #### [Read+predict in chunks](#step3)","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-warning\" style=\"font-size:14px; font-family:verdana; line-height: 1=4.7em;\">\n    \n### 📌 Steps:  \n1. ####    [Save bug df to .csv](#step1)  \n2. ####    [Reset enviroment to free RAM](#step2)  \n3. ####    [Read and predict in chunks](#step3)  \n</div>\n","metadata":{}},{"cell_type":"markdown","source":"### 1. Save big df to .csv file:<a name=\"step1\"></a>","metadata":{}},{"cell_type":"code","source":"test_data = TabularDataset(df_test.drop(columns=[\"case_id\", \"WEEK_NUM\"]))\ndel df_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.623844Z","iopub.execute_input":"2024-03-12T18:17:25.624125Z","iopub.status.idle":"2024-03-12T18:17:25.765131Z","shell.execute_reply.started":"2024-03-12T18:17:25.624101Z","shell.execute_reply":"2024-03-12T18:17:25.764246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.to_csv('test_data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.766149Z","iopub.execute_input":"2024-03-12T18:17:25.766408Z","iopub.status.idle":"2024-03-12T18:17:25.787858Z","shell.execute_reply.started":"2024-03-12T18:17:25.766386Z","shell.execute_reply":"2024-03-12T18:17:25.787169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Reset enviroment to free RAM:<a name=\"step2\"></a>\n- Note that all variables will be deleted, also you'll have to import all packages again","metadata":{}},{"cell_type":"code","source":"%reset -f\n!mkdir -p submission_data\n!mv test_data.csv submission_data/test_data.csv  # move file to submission_data dir\nimport glob\nfor f in glob.glob('*'):\n    if not f.startswith('submission_data'):# delete all files if path not starts with submission_data\n        !rm -rf {f}","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:25.789003Z","iopub.execute_input":"2024-03-12T18:17:25.789632Z","iopub.status.idle":"2024-03-12T18:17:27.956318Z","shell.execute_reply.started":"2024-03-12T18:17:25.789599Z","shell.execute_reply":"2024-03-12T18:17:27.955070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Read+predict in chunks, [reference](https://auto.gluon.ai/0.7.0/tutorials/tabular_prediction/tabular-faq.html#how-can-i-perform-inference-on-a-file-that-wont-fit-in-memory)<a name=\"step3\"></a>","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom autogluon.tabular import TabularDataset, TabularPredictor","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:27.957923Z","iopub.execute_input":"2024-03-12T18:17:27.958208Z","iopub.status.idle":"2024-03-12T18:17:27.965054Z","shell.execute_reply.started":"2024-03-12T18:17:27.958183Z","shell.execute_reply":"2024-03-12T18:17:27.964135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PATH = \"/kaggle/input/home-credit-automl-models-dataset/\"\npredictor = TabularPredictor.load(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:27.966328Z","iopub.execute_input":"2024-03-12T18:17:27.966592Z","iopub.status.idle":"2024-03-12T18:17:27.989227Z","shell.execute_reply.started":"2024-03-12T18:17:27.966570Z","shell.execute_reply":"2024-03-12T18:17:27.988534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CHUNK_SIZE = 10 ** 6\nreader = pd.read_csv('/kaggle/working/submission_data/test_data.csv', chunksize=CHUNK_SIZE)\ny_pred = []\nfor df_chunk in reader:\n    y_pred.append(predictor.predict_proba(df_chunk).iloc[:, 1].values)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:27.990150Z","iopub.execute_input":"2024-03-12T18:17:27.990422Z","iopub.status.idle":"2024-03-12T18:17:39.478046Z","shell.execute_reply.started":"2024-03-12T18:17:27.990399Z","shell.execute_reply":"2024-03-12T18:17:39.477227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = np.concatenate(y_pred, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:39.479296Z","iopub.execute_input":"2024-03-12T18:17:39.480398Z","iopub.status.idle":"2024-03-12T18:17:39.484915Z","shell.execute_reply.started":"2024-03-12T18:17:39.480361Z","shell.execute_reply":"2024-03-12T18:17:39.483975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"del reader\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:39.486239Z","iopub.execute_input":"2024-03-12T18:17:39.486573Z","iopub.status.idle":"2024-03-12T18:17:39.696491Z","shell.execute_reply.started":"2024-03-12T18:17:39.486540Z","shell.execute_reply":"2024-03-12T18:17:39.695505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:39.701046Z","iopub.execute_input":"2024-03-12T18:17:39.701343Z","iopub.status.idle":"2024-03-12T18:17:39.709653Z","shell.execute_reply.started":"2024-03-12T18:17:39.701318Z","shell.execute_reply":"2024-03-12T18:17:39.708758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())\n\ndf_subm.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:39.710848Z","iopub.execute_input":"2024-03-12T18:17:39.711161Z","iopub.status.idle":"2024-03-12T18:17:39.723592Z","shell.execute_reply.started":"2024-03-12T18:17:39.711137Z","shell.execute_reply":"2024-03-12T18:17:39.722694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-12T18:17:39.724670Z","iopub.execute_input":"2024-03-12T18:17:39.724961Z","iopub.status.idle":"2024-03-12T18:17:39.732911Z","shell.execute_reply.started":"2024-03-12T18:17:39.724938Z","shell.execute_reply":"2024-03-12T18:17:39.732101Z"},"trusted":true},"execution_count":null,"outputs":[]}]}