{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":162314401,"sourceType":"kernelVersion"},{"sourceId":162317063,"sourceType":"kernelVersion"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/autogluon-pkgs autogluon > /dev/null\n!python -m pip install --no-index --find-links=/kaggle/input/ray-pkgs --upgrade --force-reinstall -q ray==2.6.3\n# tips: https://github.com/autogluon/autogluon/issues/3365","metadata":{"execution":{"iopub.status.busy":"2024-02-29T02:32:02.427601Z","iopub.execute_input":"2024-02-29T02:32:02.428618Z","iopub.status.idle":"2024-02-29T02:36:46.355164Z","shell.execute_reply.started":"2024-02-29T02:32:02.428566Z","shell.execute_reply":"2024-02-29T02:36:46.352719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Helper Functions","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nfrom glob import glob\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport gc\nimport warnings\n\n# Ignore future warnings\nwarnings.simplefilter(action=\"ignore\", category=FutureWarning)\n\n# Define a Pipeline class for preprocessing\nclass Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        # Set data types for various columns\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    @staticmethod\n    def handle_dates(df):\n        # Convert dates to a numerical feature based on \"date_decision\"\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n        df = df.drop(\"date_decision\")\n        return df\n\n    @staticmethod\n    def filter_cols(df):\n        # Filter out columns based on missing values or lack of variance\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"] and df[col].dtype == pl.String:\n                freq = df[col].n_unique()\n                if freq == 1 or freq > 200:\n                    df = df.drop(col)\n        return df\n\n# Define an Aggregator class for feature engineering\nclass Aggregator:\n    @staticmethod\n    def get_exprs(df):\n        # Aggregate expressions for different data types\n        num_cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        date_cols = [col for col in df.columns if col[-1] in (\"D\",)]\n        str_cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        other_cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        count_cols = [col for col in df.columns if \"num_group\" in col]\n\n        exprs = ([pl.max(col).alias(f\"max_{col}\") for col in num_cols + date_cols + str_cols + other_cols + count_cols])\n        return exprs\n\n# Define functions for file I/O operations\ndef read_file(path, depth=None):\n    # Read a single file and preprocess it\n    df = pl.read_parquet(path).pipe(Pipeline.set_table_dtypes)\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    return df\n\ndef read_files(regex_path, depth=None):\n    # Read multiple files matching a pattern and preprocess them\n    chunks = [read_file(path) for path in glob(str(regex_path))]\n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    return df\n\n# Define a function for feature engineering\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    # Combine features from different depths\n    df_base = df_base.with_columns(\n        month_decision=pl.col(\"date_decision\").dt.month(),\n        weekday_decision=pl.col(\"date_decision\").dt.weekday(),\n    )\n\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base\n\n# Convert Polars DataFrame to Pandas DataFrame for further processing\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(include=['object']).columns)\n    df_data[cat_cols] = df_data[cat_cols].astype('category')\n    return df_data, cat_cols\n","metadata":{"execution":{"iopub.status.busy":"2024-02-29T02:38:18.949700Z","iopub.execute_input":"2024-02-29T02:38:18.950874Z","iopub.status.idle":"2024-02-29T02:38:20.134178Z","shell.execute_reply.started":"2024-02-29T02:38:18.950822Z","shell.execute_reply":"2024-02-29T02:38:20.132974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load and Preprocess Data","metadata":{}},{"cell_type":"code","source":"# Configuration for file paths\nROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"\n\n# Function to read and preprocess train and test datasets\ndef load_and_preprocess_data():\n    # Load train base data and feature engineer it\n    train_data_store = {\n        \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n        \"depth_0\": [\n            read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n            read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n        ],\n        \"depth_1\": [\n            read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n            read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n            read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n            read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n            read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n            read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n            read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n            read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n            read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n        ],\n        \"depth_2\": [\n            read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        ],\n    }\n\n    df_train = feature_eng(**train_data_store)\n    df_train = df_train.pipe(Pipeline.filter_cols)\n    df_train, cat_cols = to_pandas(df_train)\n    del train_data_store\n    gc.collect()\n\n    # Load test base data and feature engineer it similarly\n    test_data_store = {\n        \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n        \"depth_0\": [\n            read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n            read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n        ],\n        \"depth_1\": [\n            read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n            read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n            read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n            read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n            read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n            read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n            read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n            read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n            read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n        ],\n        \"depth_2\": [\n            read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        ],\n    }\n\n    df_test = feature_eng(**test_data_store)\n    df_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n    df_test, _ = to_pandas(df_test, cat_cols)\n    del test_data_store\n    gc.collect()\n\n    return df_train, df_test, cat_cols\n\n# Load train and test datasets\ndf_train, df_test, cat_cols = load_and_preprocess_data()","metadata":{"execution":{"iopub.status.busy":"2024-02-29T02:38:23.967805Z","iopub.execute_input":"2024-02-29T02:38:23.968436Z","iopub.status.idle":"2024-02-29T02:39:25.365515Z","shell.execute_reply.started":"2024-02-29T02:38:23.968400Z","shell.execute_reply":"2024-02-29T02:39:25.364628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train with AutoGluton","metadata":{}},{"cell_type":"code","source":"from autogluon.tabular import TabularDataset, TabularPredictor\n\ntrain_data = TabularDataset(df_train.drop(columns=[\"case_id\", \"WEEK_NUM\"]))\nlabel = \"target\"","metadata":{"execution":{"iopub.status.busy":"2024-02-21T03:51:32.248380Z","iopub.execute_input":"2024-02-21T03:51:32.248712Z","iopub.status.idle":"2024-02-21T04:01:46.294928Z","shell.execute_reply.started":"2024-02-21T03:51:32.248684Z","shell.execute_reply":"2024-02-21T04:01:46.293606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictor = TabularPredictor(label=label).fit(train_data, presets='best_quality', time_limit=36000)\npredictor.leaderboard()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Trained Model","metadata":{}},{"cell_type":"code","source":"# from autogluon.tabular import TabularDataset, TabularPredictor\n\n# train_data = TabularDataset(df_train.drop(columns=[\"case_id\", \"WEEK_NUM\"]))\n# label = \"target\"\n\n# predictor = TabularPredictor.load(path=\"/kaggle/input/automl-with-autogluton-home-credit/AutogluonModels/ag-20240221_074247\")\n# predictor.leaderboard()","metadata":{"execution":{"iopub.status.busy":"2024-02-29T02:39:31.615702Z","iopub.execute_input":"2024-02-29T02:39:31.616782Z","iopub.status.idle":"2024-02-29T02:39:35.499153Z","shell.execute_reply.started":"2024-02-29T02:39:31.616744Z","shell.execute_reply":"2024-02-29T02:39:35.498360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Prior to calibration (predictor.decision_threshold={predictor.decision_threshold}):')\nscores = predictor.evaluate(train_data)\nprint(f'\\tScores:{scores}')\n\ncalibrated_decision_threshold = predictor.calibrate_decision_threshold(metric=\"accuracy\", decision_thresholds=50)\npredictor.set_decision_threshold(calibrated_decision_threshold)\n\nprint(f'After calibration (predictor.decision_threshold={predictor.decision_threshold}):')\nscores_calibrated = predictor.evaluate(train_data)\nprint(f'\\tScores:{scores_calibrated}')","metadata":{"execution":{"iopub.status.busy":"2024-02-29T02:39:52.202826Z","iopub.execute_input":"2024-02-29T02:39:52.203441Z","iopub.status.idle":"2024-02-29T02:44:02.924346Z","shell.execute_reply.started":"2024-02-29T02:39:52.203412Z","shell.execute_reply":"2024-02-29T02:44:02.923160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare for Submission","metadata":{}},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"WEEK_NUM\"]).set_index(\"case_id\")\ny_pred = predictor.predict_proba(X_test)\n\nsubmission = pd.DataFrame({\n    \"case_id\": y_pred.index,\n    \"score\": y_pred[1].values\n})\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-21T04:01:46.495598Z","iopub.execute_input":"2024-02-21T04:01:46.495899Z","iopub.status.idle":"2024-02-21T04:01:46.506637Z","shell.execute_reply.started":"2024-02-21T04:01:46.495876Z","shell.execute_reply":"2024-02-21T04:01:46.505574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T04:01:46.507557Z","iopub.execute_input":"2024-02-21T04:01:46.507832Z","iopub.status.idle":"2024-02-21T04:01:46.521691Z","shell.execute_reply.started":"2024-02-21T04:01:46.507794Z","shell.execute_reply":"2024-02-21T04:01:46.520854Z"},"trusted":true},"execution_count":null,"outputs":[]}]}