{"metadata":{"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":33095,"sourceType":"modelInstanceVersion","modelInstanceId":27710},{"sourceId":33096,"sourceType":"modelInstanceVersion","modelInstanceId":27711}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.13"},"papermill":{"default_parameters":{},"duration":17.915251,"end_time":"2024-04-18T01:06:04.765569","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-04-18T01:05:46.850318","version":"2.5.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# *Unofficial Winning Solution - 0.618 on Private LB / 0.626 on Public LB*","metadata":{}},{"cell_type":"markdown","source":"### **Credit:** goes to @jzhllin based on his amazing post-processing sorting strategy. \n\n- https://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/discussion/507959","metadata":{}},{"cell_type":"code","source":"import joblib  \nfrom pathlib import Path  \nimport gc  \nfrom glob import glob  \nimport numpy as np  \nimport pandas as pd  \nimport polars as pl  \nfrom sklearn.base import BaseEstimator, RegressorMixin  \nfrom sklearn.metrics import roc_auc_score  \nimport lightgbm as lgb  \nimport warnings  \nwarnings.filterwarnings('ignore')  \n\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'  ","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":6.044727,"end_time":"2024-04-18T01:05:56.081208","exception":false,"start_time":"2024-04-18T01:05:50.036481","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:22.981729Z","iopub.execute_input":"2024-05-28T06:36:22.982332Z","iopub.status.idle":"2024-05-28T06:36:28.361550Z","shell.execute_reply.started":"2024-05-28T06:36:22.982301Z","shell.execute_reply":"2024-05-28T06:36:28.360752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  \n                df = df.with_columns(pl.col(col).dt.total_days())  \n        df = df.drop(\"date_decision\", \"MONTH\")  \n        return df\n\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.7:\n                    df = df.drop(col)  \n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)  \n        \n        return df","metadata":{"papermill":{"duration":0.034893,"end_time":"2024-04-18T01:05:56.121806","exception":false,"start_time":"2024-04-18T01:05:56.086913","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:28.363226Z","iopub.execute_input":"2024-05-28T06:36:28.363488Z","iopub.status.idle":"2024-05-28T06:36:28.375404Z","shell.execute_reply.started":"2024-05-28T06:36:28.363465Z","shell.execute_reply":"2024-05-28T06:36:28.374473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]  \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]  \n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols] \n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]  \n        expr_median = [pl.median(col).alias(f\"median_{col}\") for col in cols]  \n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols] \n        return expr_max + expr_last + expr_mean \n\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\")]  \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]  \n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]  \n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]  \n        expr_median = [pl.median(col).alias(f\"median_{col}\") for col in cols]  \n        return expr_max + expr_last + expr_mean \n\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]  \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]  \n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]  \n        return expr_max + expr_last\n\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]  \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]  \n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]  \n        return expr_max + expr_last\n\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col] \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]  \n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]  \n        return expr_max + expr_last\n\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-28T06:36:28.376818Z","iopub.execute_input":"2024-05-28T06:36:28.377297Z","iopub.status.idle":"2024-05-28T06:36:28.395096Z","shell.execute_reply.started":"2024-05-28T06:36:28.377263Z","shell.execute_reply":"2024-05-28T06:36:28.394330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df)) \n    return df","metadata":{"papermill":{"duration":0.032096,"end_time":"2024-04-18T01:05:56.159326","exception":false,"start_time":"2024-04-18T01:05:56.12723","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:28.397196Z","iopub.execute_input":"2024-05-28T06:36:28.397450Z","iopub.status.idle":"2024-05-28T06:36:28.409993Z","shell.execute_reply.started":"2024-05-28T06:36:28.397428Z","shell.execute_reply":"2024-05-28T06:36:28.409287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        chunks.append(df)\n    df = pl.concat(chunks, how=\"vertical_relaxed\").unique(subset=[\"case_id\"])\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-28T06:36:28.411021Z","iopub.execute_input":"2024-05-28T06:36:28.411273Z","iopub.status.idle":"2024-05-28T06:36:28.421141Z","shell.execute_reply.started":"2024-05-28T06:36:28.411251Z","shell.execute_reply":"2024-05-28T06:36:28.420239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = df_base.with_columns(\n        month_decision = pl.col(\"date_decision\").dt.month(),\n        weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-28T06:36:28.422260Z","iopub.execute_input":"2024-05-28T06:36:28.422514Z","iopub.status.idle":"2024-05-28T06:36:28.432209Z","shell.execute_reply.started":"2024-05-28T06:36:28.422492Z","shell.execute_reply":"2024-05-28T06:36:28.431338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-28T06:36:28.433264Z","iopub.execute_input":"2024-05-28T06:36:28.433540Z","iopub.status.idle":"2024-05-28T06:36:28.442176Z","shell.execute_reply.started":"2024-05-28T06:36:28.433518Z","shell.execute_reply":"2024-05-28T06:36:28.441391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" \n    Iterate through all the columns of a dataframe and modify the data type\n    to reduce memory usage.\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2  # Memory usage before optimization\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2  # Memory usage after optimization\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-28T06:36:28.443208Z","iopub.execute_input":"2024-05-28T06:36:28.443486Z","iopub.status.idle":"2024-05-28T06:36:28.456638Z","shell.execute_reply.started":"2024-05-28T06:36:28.443463Z","shell.execute_reply":"2024-05-28T06:36:28.455762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_notebook_info = joblib.load('/kaggle/input/homecredit-models-public/other/lgb/1/notebook_info.joblib')\n\n# Print notebook information\nprint(f\"- [lgb] notebook_start_time: {lgb_notebook_info['notebook_start_time']}\")\nprint(f\"- [lgb] description: {lgb_notebook_info['description']}\")\n\n# Load columns and categorical columns\ncols = lgb_notebook_info['cols']\ncat_cols = lgb_notebook_info['cat_cols']\nprint(f\"- [lgb] len(cols): {len(cols)}\")\nprint(f\"- [lgb] len(cat_cols): {len(cat_cols)}\")\n\n# Load LightGBM models\nlgb_models = joblib.load('/kaggle/input/homecredit-models-public/other/lgb/1/lgb_models.joblib')\nlgb_models","metadata":{"papermill":{"duration":0.722458,"end_time":"2024-04-18T01:05:56.898184","exception":false,"start_time":"2024-04-18T01:05:56.175726","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:28.457681Z","iopub.execute_input":"2024-05-28T06:36:28.457998Z","iopub.status.idle":"2024-05-28T06:36:29.102585Z","shell.execute_reply.started":"2024-05-28T06:36:28.457967Z","shell.execute_reply":"2024-05-28T06:36:29.101698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load categorical model notebook information\ncat_notebook_info = joblib.load('/kaggle/input/homecredit-models-public/other/cat/1/notebook_info.joblib')\n\n# Print notebook information\nprint(f\"- [cat] notebook_start_time: {cat_notebook_info['notebook_start_time']}\")\nprint(f\"- [cat] description: {cat_notebook_info['description']}\")\n\n# Load categorical models\ncat_models = joblib.load('/kaggle/input/homecredit-models-public/other/cat/1/cat_models.joblib')\ncat_models","metadata":{"papermill":{"duration":4.878543,"end_time":"2024-04-18T01:06:01.784082","exception":false,"start_time":"2024-04-18T01:05:56.905539","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:29.105307Z","iopub.execute_input":"2024-05-28T06:36:29.105594Z","iopub.status.idle":"2024-05-28T06:36:33.720982Z","shell.execute_reply.started":"2024-05-28T06:36:29.105570Z","shell.execute_reply":"2024-05-28T06:36:33.720063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the root directory path\nROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n\n# Define the directory path for the test data\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"\n\n# Create a dictionary to store different dataframes generated from reading parquet files\ndata_store = {\n    # Read the base test data and store it with the key 'df_base'\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \n    # Read depth 0 data, which includes static data and additional files matching a pattern\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \n    # Read depth 1 data, including various files related to applicant previous applications, tax registries,\n    # credit bureau data, and other information\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \n    # Read depth 2 data, which includes additional credit bureau data, applicant previous applications,\n    # and personal information\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}","metadata":{"papermill":{"duration":0.545928,"end_time":"2024-04-18T01:06:02.349483","exception":false,"start_time":"2024-04-18T01:06:01.803555","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:33.722547Z","iopub.execute_input":"2024-05-28T06:36:33.722856Z","iopub.status.idle":"2024-05-28T06:36:34.170584Z","shell.execute_reply.started":"2024-05-28T06:36:33.722830Z","shell.execute_reply":"2024-05-28T06:36:34.169734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform feature engineering on the test data using the provided data store\ndf_test = feature_eng(**data_store)\n\n# Print the shape of the test data before further processing\nprint(\"test data shape:\\t\", df_test.shape)\n\n# Clean up memory by deleting the data store and running garbage collection\ndel data_store\ngc.collect()\n\n# Select columns of interest from the test data\ndf_test = df_test.select(['case_id'] + cols)\n\n# Convert the test data to a pandas DataFrame and optimize memory usage\ndf_test, cat_cols = to_pandas(df_test, cat_cols)\ndf_test = reduce_mem_usage(df_test)\n\n# Set the case_id column as the index of the DataFrame\ndf_test = df_test.set_index('case_id')\n\n# Print the shape of the test data after processing\nprint(\"test data shape:\\t\", df_test.shape)\n\n# Run garbage collection to clean up memory\ngc.collect()","metadata":{"papermill":{"duration":0.654075,"end_time":"2024-04-18T01:06:03.010344","exception":false,"start_time":"2024-04-18T01:06:02.356269","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:34.171717Z","iopub.execute_input":"2024-05-28T06:36:34.172022Z","iopub.status.idle":"2024-05-28T06:36:34.732775Z","shell.execute_reply.started":"2024-05-28T06:36:34.171997Z","shell.execute_reply":"2024-05-28T06:36:34.731800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"papermill":{"duration":0.05776,"end_time":"2024-04-18T01:06:03.074529","exception":false,"start_time":"2024-04-18T01:06:03.016769","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:34.734011Z","iopub.execute_input":"2024-05-28T06:36:34.734300Z","iopub.status.idle":"2024-05-28T06:36:34.777671Z","shell.execute_reply.started":"2024-05-28T06:36:34.734275Z","shell.execute_reply":"2024-05-28T06:36:34.776637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        \"\"\"\n        Fit the VotingModel.\n        \n        Parameters:\n        - X: array-like or sparse matrix of shape (n_samples, n_features)\n            The input samples.\n        - y: array-like of shape (n_samples,), default=None\n            The target values.\n            \n        Returns:\n        - self: object\n            Returns self.\n        \"\"\"\n        return self\n    \n    def predict(self, X):\n        \"\"\"\n        Predict regression target for X.\n        \n        Parameters:\n        - X: array-like or sparse matrix of shape (n_samples, n_features)\n            The input samples.\n            \n        Returns:\n        - y_preds: array-like of shape (n_samples,)\n            The predicted target values.\n        \"\"\"\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n     \n    def predict_proba(self, X):      \n        \"\"\"\n        Predict class probabilities for X.\n        \n        Parameters:\n        - X: array-like or sparse matrix of shape (n_samples, n_features)\n            The input samples.\n            \n        Returns:\n        - proba: array-like of shape (n_samples, n_classes)\n            Class probabilities of the input samples.\n        \"\"\"\n        # lgb\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators[:5]]\n        \n        # cat        \n        X[cat_cols] = X[cat_cols].astype(str)\n        y_preds += [estimator.predict_proba(X) for estimator in self.estimators[-5:]]\n        \n        return np.mean(y_preds, axis=0)","metadata":{"papermill":{"duration":0.019084,"end_time":"2024-04-18T01:06:03.118007","exception":false,"start_time":"2024-04-18T01:06:03.098923","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:34.779228Z","iopub.execute_input":"2024-05-28T06:36:34.779879Z","iopub.status.idle":"2024-05-28T06:36:34.788876Z","shell.execute_reply.started":"2024-05-28T06:36:34.779842Z","shell.execute_reply":"2024-05-28T06:36:34.787973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = VotingModel(lgb_models + cat_models)\nlen(model.estimators)","metadata":{"papermill":{"duration":0.015923,"end_time":"2024-04-18T01:06:03.142213","exception":false,"start_time":"2024-04-18T01:06:03.12629","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-05-28T06:36:34.790089Z","iopub.execute_input":"2024-05-28T06:36:34.790371Z","iopub.status.idle":"2024-05-28T06:36:34.803777Z","shell.execute_reply.started":"2024-05-28T06:36:34.790348Z","shell.execute_reply":"2024-05-28T06:36:34.802869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Finally, add the following Post-processing according to** @jzhllin **amazing sorting strategy.** \n\n- https://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/discussion/507959","metadata":{}},{"cell_type":"code","source":"def read_files2(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        df = df.select(['case_id', 'dpdmaxdateyear_596T', 'dpdmaxdatemonth_89T',\n                        'overdueamountmaxdateyear_2T','overdueamountmaxdatemonth_365T','num_group1'])\n        df = df.filter(df['num_group1'] == 0)\n        df = df.drop(['num_group1'])\n        chunks.append(df)\n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df\n\ndf_base =  read_file(TEST_DIR / \"test_base.parquet\")\ntmp = read_files2(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\")\ntmp = df_base.join(tmp, how=\"left\", on=\"case_id\")\ntmp = tmp.select(['case_id', 'dpdmaxdateyear_596T', 'dpdmaxdatemonth_89T',\n                  'overdueamountmaxdateyear_2T','overdueamountmaxdatemonth_365T']).to_pandas()\n\ny_pred = pd.Series(model.predict_proba(df_test)[:, 1], index=df_test.index)\n\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\ndf_subm[\"score\"] = y_pred\n\ndf_subm = df_subm.merge(tmp,how='left',on=['case_id'])\n\nyear_596T_ratio = df_subm['dpdmaxdateyear_596T'].isnull().sum()/ len(df_subm)\nyear_2T_ratio = df_subm['overdueamountmaxdateyear_2T'].isnull().sum()/ len(df_subm)\n\nif year_596T_ratio <= year_2T_ratio:\n    df_subm = df_subm.sort_values(by=['dpdmaxdateyear_596T', 'dpdmaxdatemonth_89T'])\nelse:\n    df_subm = df_subm.sort_values(by=['overdueamountmaxdateyear_2T', 'overdueamountmaxdatemonth_365T'])\n\ndf_subm = df_subm.reset_index(drop=True)\nscore_col_index = df_subm.columns.get_loc('score')\ncut_off_index = int(len(df_subm) * 0.35)\ndf_subm.iloc[:cut_off_index, score_col_index] = (df_subm.iloc[:cut_off_index, score_col_index] - 0.055).clip(0)\ndf_subm= df_subm.drop(['dpdmaxdateyear_596T','dpdmaxdatemonth_89T',\n                      'overdueamountmaxdateyear_2T','overdueamountmaxdatemonth_365T'],axis=1)\ndf_subm = df_subm.set_index(\"case_id\")\ndf_subm.to_csv(\"submission.csv\")\ndf_subm                            ","metadata":{"execution":{"iopub.status.busy":"2024-05-28T06:36:34.804724Z","iopub.execute_input":"2024-05-28T06:36:34.805040Z","iopub.status.idle":"2024-05-28T06:36:35.431619Z","shell.execute_reply.started":"2024-05-28T06:36:34.805010Z","shell.execute_reply":"2024-05-28T06:36:35.430719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.007392,"end_time":"2024-04-18T01:06:03.836532","exception":false,"start_time":"2024-04-18T01:06:03.82914","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}