{"metadata":{"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7610867,"sourceType":"datasetVersion","datasetId":4431663}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.13"},"papermill":{"default_parameters":{},"duration":888.861714,"end_time":"2024-02-08T13:46:18.589602","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-02-08T13:31:29.727888","version":"2.5.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction","metadata":{"_uuid":"1b68f4e3-4ba9-42ef-b5cc-a6cf738802f6","_cell_guid":"5b92ac5f-5784-4444-b9c3-325bd62e30dd","trusted":true}},{"cell_type":"markdown","source":"**Based on:**\n\nhttps://www.kaggle.com/code/greysky/home-credit-baseline\n\nChanges:\n1. Removed model training code (contains only data collection and preproccessing; utility script)\n1. Added function prepare_df\n1. Added new aggregations: min, mean, mode, first, last, n_unique\n\n\n**Related notebooks**\n\nTraining model-1 notebook:\n\nhttps://www.kaggle.com/andreynesterov/home-credit-baseline-training\n\nInference notebook:\n\nhttps://www.kaggle.com/andreynesterov/home-credit-baseline-inference","metadata":{"_uuid":"bb378d4e-f1f7-498f-ab20-7a2cd60c4bc5","_cell_guid":"47aa008c-1bf4-4bb7-a785-5b37abebeb36","trusted":true}},{"cell_type":"markdown","source":"# Dependencies","metadata":{"_uuid":"2197dfdb-b01d-443d-a7b7-d3d74e58a294","_cell_guid":"edcbf522-73f3-488c-98c1-564b52c244c0","trusted":true}},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', 500)","metadata":{"_uuid":"812ee031-a3f7-44c1-a4c4-fee017cb2810","_cell_guid":"e3cf6e90-28a7-4e4a-9aa9-8f592748aea2","collapsed":false,"papermill":{"duration":6.169561,"end_time":"2024-02-08T13:31:38.616695","exception":false,"start_time":"2024-02-08T13:31:32.447134","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-03-16T07:59:07.327046Z","iopub.execute_input":"2024-03-16T07:59:07.327835Z","iopub.status.idle":"2024-03-16T07:59:08.500599Z","shell.execute_reply.started":"2024-03-16T07:59:07.327803Z","shell.execute_reply":"2024-03-16T07:59:08.499851Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{"_uuid":"a39ecbe4-3a14-4701-be2b-0995c72839ae","_cell_guid":"4ab674cf-9f74-4f34-bee6-c9a965d1ef5c","papermill":{"duration":0.008354,"end_time":"2024-02-08T13:31:38.872593","exception":false,"start_time":"2024-02-08T13:31:38.864239","status":"completed"},"tags":[],"trusted":true}},{"cell_type":"code","source":"class CFG:\n    root_dir = Path(\"/kaggle/input/home-credit-credit-risk-model-stability/\")\n    train_dir = Path(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/train\")\n    test_dir = Path(\"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files/test\")","metadata":{"_uuid":"8a44fb01-ffe7-4639-821e-028842eaf8f4","_cell_guid":"46595397-ee69-49f2-80ee-3cd4bf4858db","collapsed":false,"papermill":{"duration":0.01558,"end_time":"2024-02-08T13:31:38.89682","exception":false,"start_time":"2024-02-08T13:31:38.88124","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-03-16T07:59:08.502321Z","iopub.execute_input":"2024-03-16T07:59:08.502686Z","iopub.status.idle":"2024-03-16T07:59:08.508144Z","shell.execute_reply.started":"2024-03-16T07:59:08.502656Z","shell.execute_reply":"2024-03-16T07:59:08.507200Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature definitions","metadata":{"_uuid":"9fa93b24-0f94-47d8-a0ee-2278a0761d97","_cell_guid":"e8f74706-c7dd-47d9-af3a-4e8b9f98262a","trusted":true}},{"cell_type":"code","source":"if __name__ == '__main__':\n    feature_definitions_df = pd.read_csv(CFG.root_dir / \"feature_definitions.csv\")\n    display(feature_definitions_df)\n    pd.reset_option(\"display.max_rows\", 0)","metadata":{"_uuid":"c5d99226-a339-4be4-9b9a-ad9fb9c22461","_cell_guid":"9ca49470-37dc-4f19-a74b-5c124766ae62","collapsed":false,"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-03-16T07:59:08.509235Z","iopub.execute_input":"2024-03-16T07:59:08.509491Z","iopub.status.idle":"2024-03-16T07:59:08.587917Z","shell.execute_reply.started":"2024-03-16T07:59:08.509468Z","shell.execute_reply":"2024-03-16T07:59:08.587005Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Collection and Preprocessing","metadata":{"_uuid":"2bdd65c0-e3d2-4a2d-9f64-fd6d1abb58d7","_cell_guid":"25fa4fc4-20b5-4036-85a7-cf591e844dac","trusted":true}},{"cell_type":"markdown","source":"### Pipeline","metadata":{"_uuid":"c5a4c7d2-757e-4d4f-ba65-f7687082ce98","_cell_guid":"12fe4407-4c34-44fe-b595-244283e76346","papermill":{"duration":0.007632,"end_time":"2024-02-08T13:31:38.673586","exception":false,"start_time":"2024-02-08T13:31:38.665954","status":"completed"},"tags":[],"trusted":true}},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"_uuid":"d08841a6-b3cb-4de3-a665-512a8390fe1b","_cell_guid":"f323e01b-e9e2-4efc-80fc-6f31d4d5b08f","collapsed":false,"papermill":{"duration":0.022599,"end_time":"2024-02-08T13:31:38.704022","exception":false,"start_time":"2024-02-08T13:31:38.681423","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-03-16T07:59:08.589051Z","iopub.execute_input":"2024-03-16T07:59:08.589327Z","iopub.status.idle":"2024-03-16T07:59:08.601633Z","shell.execute_reply.started":"2024-03-16T07:59:08.589302Z","shell.execute_reply":"2024-03-16T07:59:08.600869Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Automatic Aggregation","metadata":{"_uuid":"62d012e9-0204-467b-8ce7-fc4fe95c2b33","_cell_guid":"f05efead-e560-47da-bc8d-0f4333ddb465","papermill":{"duration":0.00774,"end_time":"2024-02-08T13:31:38.719798","exception":false,"start_time":"2024-02-08T13:31:38.712058","status":"completed"},"tags":[],"trusted":true}},{"cell_type":"code","source":"class Aggregator:\n    num_aggregators = [pl.max, pl.min, pl.first, pl.last, pl.mean]\n    str_aggregators = [pl.max, pl.min, pl.first, pl.last] # n_unique\n    group_aggregators = [pl.max, pl.min, pl.first, pl.last]\n    \n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_all = []\n        for method in Aggregator.num_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]\n            expr_all += expr\n\n        return expr_all\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n        expr_all = []\n        for method in Aggregator.num_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]  \n            expr_all += expr\n\n        return expr_all\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_all = []\n        for method in Aggregator.str_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]  \n            expr_all += expr\n            \n        expr_mode = [\n            pl.col(col)\n            .drop_nulls()\n            .mode()\n            .first()\n            .alias(f\"mode_{col}\")\n            for col in cols\n        ]\n\n        return expr_all + expr_mode\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_all = []\n        for method in Aggregator.str_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]  \n            expr_all += expr\n\n        return expr_all\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_all = []\n        for method in Aggregator.group_aggregators:\n            expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in cols]  \n            expr_all += expr\n            \n#         if len(cols) > 0:\n#             method = pl.count\n#             expr = [method(col).alias(f\"{method.__name__}_{col}\") for col in [cols[0]]]\n#             expr_all += expr\n\n        return expr_all\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"_uuid":"6f5f3114-601d-433a-8f6e-14f11eb8058e","_cell_guid":"b4ee12c2-727d-4ee3-92a9-deb461e5b68e","collapsed":false,"papermill":{"duration":0.021321,"end_time":"2024-02-08T13:31:38.750003","exception":false,"start_time":"2024-02-08T13:31:38.728682","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-03-16T07:59:08.604974Z","iopub.execute_input":"2024-03-16T07:59:08.605423Z","iopub.status.idle":"2024-03-16T07:59:08.620390Z","shell.execute_reply.started":"2024-03-16T07:59:08.605391Z","shell.execute_reply":"2024-03-16T07:59:08.619580Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### File I/O","metadata":{"_uuid":"6278db23-bb38-4e1f-9249-6714be8ea455","_cell_guid":"93e01904-a2f6-48c3-9edd-8c3295804590","papermill":{"duration":0.007841,"end_time":"2024-02-08T13:31:38.76613","exception":false,"start_time":"2024-02-08T13:31:38.758289","status":"completed"},"tags":[],"trusted":true}},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(pl.read_parquet(path).pipe(Pipeline.set_table_dtypes))\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df","metadata":{"_uuid":"46e76e1d-3944-42ec-aafc-67e8b4ae45b3","_cell_guid":"55c84f42-f70d-4d46-b193-23f4070b443b","collapsed":false,"papermill":{"duration":0.017149,"end_time":"2024-02-08T13:31:38.791428","exception":false,"start_time":"2024-02-08T13:31:38.774279","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-03-16T07:59:08.621372Z","iopub.execute_input":"2024-03-16T07:59:08.621640Z","iopub.status.idle":"2024-03-16T07:59:08.632023Z","shell.execute_reply.started":"2024-03-16T07:59:08.621616Z","shell.execute_reply":"2024-03-16T07:59:08.631243Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{"_uuid":"e850900a-a308-429b-8db1-be7927cd6e4c","_cell_guid":"6c73b386-7cef-4f5e-8bef-a57b02151da8","papermill":{"duration":0.007901,"end_time":"2024-02-08T13:31:38.807502","exception":false,"start_time":"2024-02-08T13:31:38.799601","status":"completed"},"tags":[],"trusted":true}},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"_uuid":"c41a68c2-a884-48b2-bb87-64de07e23552","_cell_guid":"5af8d342-aa89-4cf0-a11c-def51bbe75ed","collapsed":false,"papermill":{"duration":0.016482,"end_time":"2024-02-08T13:31:38.832632","exception":false,"start_time":"2024-02-08T13:31:38.81615","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-03-16T07:59:08.633101Z","iopub.execute_input":"2024-03-16T07:59:08.633628Z","iopub.status.idle":"2024-03-16T07:59:08.641090Z","shell.execute_reply.started":"2024-03-16T07:59:08.633598Z","shell.execute_reply":"2024-03-16T07:59:08.640094Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data","metadata":{"_uuid":"b638b47a-8b21-4ab5-a9c3-7fb27ddf7a5a","_cell_guid":"96428045-8b43-4bc8-b684-06358191011d","collapsed":false,"papermill":{"duration":0.015339,"end_time":"2024-02-08T13:31:38.85625","exception":false,"start_time":"2024-02-08T13:31:38.840911","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-03-16T07:59:08.642095Z","iopub.execute_input":"2024-03-16T07:59:08.642356Z","iopub.status.idle":"2024-03-16T07:59:08.649490Z","shell.execute_reply.started":"2024-03-16T07:59:08.642335Z","shell.execute_reply":"2024-03-16T07:59:08.648744Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### from https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts\n\ndef reduce_mem_usage(df, float16_as32=True):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    if float16_as32:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float16)                    \n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"_uuid":"67cedbfa-a379-4e1d-8c19-0d3f1875def9","_cell_guid":"90cec974-17e2-4c25-a487-e8d9eb4e7954","collapsed":false,"execution":{"iopub.status.busy":"2024-03-16T07:59:08.650406Z","iopub.execute_input":"2024-03-16T07:59:08.650644Z","iopub.status.idle":"2024-03-16T07:59:08.664455Z","shell.execute_reply.started":"2024-03-16T07:59:08.650622Z","shell.execute_reply":"2024-03-16T07:59:08.663623Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare df","metadata":{"_uuid":"4154fe4f-b89f-4359-b32d-ce446dc5c308","_cell_guid":"74ac1fdc-4817-4af6-893c-294d1d03fd50","papermill":{"duration":0.007961,"end_time":"2024-02-08T13:31:38.913224","exception":false,"start_time":"2024-02-08T13:31:38.905263","status":"completed"},"tags":[],"trusted":true}},{"cell_type":"code","source":"def prepare_df(data_dir, cat_cols=None, mode=\"train\", display_store=False, train_cols=[]):\n    print(\"Collecting data...\")\n    data_store = {\n        \"df_base\": read_file(data_dir / f\"{mode}_base.parquet\"),\n        \"depth_0\": [\n            read_file(data_dir / f\"{mode}_static_cb_0.parquet\"),\n            read_files(data_dir / f\"{mode}_static_0_*.parquet\"),\n        ],\n        \"depth_1\": [\n            read_files(data_dir / f\"{mode}_applprev_1_*.parquet\", 1),\n            read_file(data_dir / f\"{mode}_tax_registry_a_1.parquet\", 1),\n            read_file(data_dir / f\"{mode}_tax_registry_b_1.parquet\", 1),\n            read_file(data_dir / f\"{mode}_tax_registry_c_1.parquet\", 1),\n            read_file(data_dir / f\"{mode}_credit_bureau_b_1.parquet\", 1),\n            read_file(data_dir / f\"{mode}_other_1.parquet\", 1),\n            read_file(data_dir / f\"{mode}_person_1.parquet\", 1),\n            read_file(data_dir / f\"{mode}_deposit_1.parquet\", 1),\n            read_file(data_dir / f\"{mode}_debitcard_1.parquet\", 1),\n        ],\n        \"depth_2\": [\n            read_file(data_dir / f\"{mode}_credit_bureau_b_2.parquet\", 2),\n        ]\n    }\n    if display_store:\n        display(data_store)\n    \n    print(\"Feature engeneering...\")\n    feats_df = feature_eng(**data_store)\n    print(\"  feats_df shape:\\t\", feats_df.shape)\n    \n    del data_store\n    gc.collect()\n    \n    print(\"Filter cols...\")\n    if mode == \"train\":\n        feats_df = feats_df.pipe(Pipeline.filter_cols)\n    else:\n        train_cols = feats_df.columns if len(train_cols) == 0 else train_cols\n        feats_df = feats_df.select([col for col in train_cols if col != \"target\"])\n    print(\"  feats_df shape:\\t\", feats_df.shape)\n    \n    print(\"Convert to pandas...\")\n    feats_df = to_pandas(feats_df, cat_cols)\n    return feats_df","metadata":{"_uuid":"3cb7daa2-4bd8-4154-9fcf-9072cd1a03b8","_cell_guid":"75a2c1ed-ff3d-42b1-9589-8780ac70a0f0","collapsed":false,"execution":{"iopub.status.busy":"2024-03-16T07:59:08.665536Z","iopub.execute_input":"2024-03-16T07:59:08.665832Z","iopub.status.idle":"2024-03-16T07:59:08.676372Z","shell.execute_reply.started":"2024-03-16T07:59:08.665809Z","shell.execute_reply":"2024-03-16T07:59:08.675396Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    train_df = prepare_df(CFG.train_dir)\n    cat_cols = list(train_df.select_dtypes(\"category\").columns)","metadata":{"_uuid":"20e16fd1-7f73-41c7-be74-b38e16c6ec22","_cell_guid":"b8babd6b-2a5e-44d9-8003-ae72ca345805","collapsed":false,"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-03-16T07:59:08.677475Z","iopub.execute_input":"2024-03-16T07:59:08.677894Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    display(train_df)","metadata":{"_uuid":"416afd6d-dedf-4617-b36e-28eff59d4818","_cell_guid":"2c0cae7f-cd23-484c-af57-106c80b5ea2c","collapsed":false,"_kg_hide-output":true,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    display(cat_cols)","metadata":{"_uuid":"f73668a1-4752-4ca6-b519-52fae6f283b6","_cell_guid":"aa8aaf7c-f45f-4324-a700-ccc77ed3a93b","collapsed":false,"_kg_hide-output":true,"scrolled":true,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    test_df = prepare_df(CFG.test_dir, cat_cols=cat_cols, mode=\"test\", train_cols=train_df.columns)","metadata":{"_uuid":"c51ceb5a-bebc-4a06-a206-6f02495fb2cd","_cell_guid":"50bb0b1f-3876-46a1-8f7f-195cf1c76997","collapsed":false,"_kg_hide-output":true,"scrolled":true,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    display(test_df)","metadata":{"_uuid":"05c610a2-eaa3-44ac-87db-c0f805de5630","_cell_guid":"2cf796ea-d16e-4566-958f-9816a997faed","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reduce memory usage and save","metadata":{"_uuid":"318b359c-b7d6-4776-90d4-aec2b70d1258","_cell_guid":"6b934d0a-368f-47c2-9b1a-79cb5ba31446","trusted":true}},{"cell_type":"code","source":"if __name__ == '__main__':\n    train_df = reduce_mem_usage(train_df)\n    test_df = reduce_mem_usage(test_df)","metadata":{"_uuid":"92545029-7dcb-4205-9371-55ad2148bd28","_cell_guid":"0c31a59e-59a3-4d79-b355-435ec8f426d3","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    train_df.to_parquet(\"train_full.parquet\")","metadata":{"_uuid":"9b61502b-a765-4a73-bb6a-0f611e014cca","_cell_guid":"847a9d80-14c6-4d13-b571-2373e32f770b","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA","metadata":{"_uuid":"ef0056df-c1c2-48cb-a256-118a15cb8c11","_cell_guid":"4efd409a-6e15-4c86-be9b-dd38c2f226e2","papermill":{"duration":0.008701,"end_time":"2024-02-08T13:32:31.919166","exception":false,"start_time":"2024-02-08T13:32:31.910465","status":"completed"},"tags":[],"trusted":true}},{"cell_type":"code","source":"if __name__ == '__main__':\n    print(\"Train is duplicated:\\t\", train_df[\"case_id\"].duplicated().any())\n    print(\"Train Week Range:\\t\", (train_df[\"WEEK_NUM\"].min(), train_df[\"WEEK_NUM\"].max()))\n\n    print()\n\n    print(\"Test is duplicated:\\t\", test_df[\"case_id\"].duplicated().any())\n    print(\"Test Week Range:\\t\", (test_df[\"WEEK_NUM\"].min(), test_df[\"WEEK_NUM\"].max()))","metadata":{"_uuid":"3cead602-92ce-4b73-9751-698848ceb0ab","_cell_guid":"2e51368b-0c62-46ba-b140-a5514a54ef93","collapsed":false,"papermill":{"duration":0.050383,"end_time":"2024-02-08T13:32:31.9784","exception":false,"start_time":"2024-02-08T13:32:31.928017","status":"completed"},"tags":[],"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    sns.lineplot(\n        data=train_df,\n        x=\"WEEK_NUM\",\n        y=\"target\",\n    )\n    plt.show()","metadata":{"_uuid":"8bcf9728-f382-4a29-8a7e-a9bddf45107b","_cell_guid":"0afeb7ef-eda1-4f44-986e-b4818a61cb2f","collapsed":false,"papermill":{"duration":16.920109,"end_time":"2024-02-08T13:32:48.907503","exception":false,"start_time":"2024-02-08T13:32:31.987394","status":"completed"},"tags":[],"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}