{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":7830825,"sourceType":"datasetVersion","datasetId":4589314},{"sourceId":172824708,"sourceType":"kernelVersion"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.metrics import roc_auc_score \nimport pickle\nimport matplotlib.pyplot as plt\nimport os \nfrom gc import collect\nfrom os import path, walk, getpid\nfrom psutil import Process\nimport ctypes;\nlibc = ctypes.CDLL(\"libc.so.6\");","metadata":{"execution":{"iopub.status.busy":"2024-04-24T09:12:05.851017Z","iopub.execute_input":"2024-04-24T09:12:05.851297Z","iopub.status.idle":"2024-04-24T09:12:11.253073Z","shell.execute_reply.started":"2024-04-24T09:12:05.851272Z","shell.execute_reply":"2024-04-24T09:12:11.252295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    experiment_number = 1\n    version = 6\n    num_folds = 5\n    mode = 'train'\n    data_path = \"/kaggle/input/home-credit-credit-risk-model-stability/\"\n    train_base_path = \"/kaggle/input/folded-base-train/train_base_week_folds.csv\"\n    train_path         = \"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train\";\n    test_path          = \"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test\";\n    null_cut_off = 0.96\n    covid_date = \"2020-02-29\"","metadata":{"execution":{"iopub.status.busy":"2024-04-24T09:40:18.237372Z","iopub.execute_input":"2024-04-24T09:40:18.238250Z","iopub.status.idle":"2024-04-24T09:40:18.243389Z","shell.execute_reply.started":"2024-04-24T09:40:18.238217Z","shell.execute_reply":"2024-04-24T09:40:18.242425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Utilities:\n    @staticmethod\n    def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64));\n            elif len(col) == 0:\n                df = df.drop(col)\n            elif col in [\"date_decision\"] or col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date));\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64));\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String));\n\n        return df;\n\n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > Config.null_cut_off:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n        return df \n    @staticmethod\n    def CleanMemory():\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\";\n\n        collect();\n        libc.malloc_trim(0);\n        pid        = getpid();\n        py         = Process(pid);\n        memory_use = py.memory_info()[0] / 2. ** 30;\n        return f\"RAM usage = {memory_use :.4} GB\";","metadata":{"execution":{"iopub.status.busy":"2024-04-24T09:40:19.156953Z","iopub.execute_input":"2024-04-24T09:40:19.157302Z","iopub.status.idle":"2024-04-24T09:40:19.172691Z","shell.execute_reply.started":"2024-04-24T09:40:19.157269Z","shell.execute_reply":"2024-04-24T09:40:19.171680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-04-24T09:40:21.158603Z","iopub.execute_input":"2024-04-24T09:40:21.159020Z","iopub.status.idle":"2024-04-24T09:40:21.170089Z","shell.execute_reply.started":"2024-04-24T09:40:21.158989Z","shell.execute_reply":"2024-04-24T09:40:21.169121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef read_file(path, depth=None):\n    df = pl.read_csv(path).pipe(Utilities.set_table_dtypes)\n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    return df \n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        chunks.append(read_file(path, depth))\n    df = pl.concat(chunks, how='vertical_relaxed')\n    df = df.unique(subset=[\"case_id\"])\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-24T09:40:21.621280Z","iopub.execute_input":"2024-04-24T09:40:21.621644Z","iopub.status.idle":"2024-04-24T09:40:21.628165Z","shell.execute_reply.started":"2024-04-24T09:40:21.621616Z","shell.execute_reply":"2024-04-24T09:40:21.627253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-24T10:10:00.850636Z","iopub.execute_input":"2024-04-24T10:10:00.851264Z","iopub.status.idle":"2024-04-24T10:10:00.881388Z","shell.execute_reply.started":"2024-04-24T10:10:00.851232Z","shell.execute_reply":"2024-04-24T10:10:00.880410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataTransformer:\n    def __init__(self, selected_columns=[], selected_cat_columns=[]):\n        self.mode = Config.mode.lower()\n        self.path = Config.train_path if self.mode == 'train' else Config.test_path\n        self.selected_columns = selected_columns \n        self.selected_cat_columns = selected_cat_columns\n\n    \n    def transform_base(self, base):\n        base = base.with_columns(pl.col('date_decision').cast(pl.Date))\n        base = base.with_columns((pl.col('date_decision') > pd.to_datetime(Config.covid_date)).alias('covid'))   \n        return base \n\n    def transform_static(self, static):\n        with open('/kaggle/input/hc-static-preprocessing/static_cols_to_drop_V3.pkl', 'rb') as f:\n            cols_to_drop = pickle.load(f)\n        static = static.drop(cols_to_drop)\n        cols_to_cast = [c for c in static.columns if (c[-1] == 'L') and (c[:3] == 'num')]\n        static = static.with_columns(pl.col(cols_to_cast).cast(pl.Float64))\n        return static\n    \n    def transform_cb(self, cb):\n#         cb = cb.join(base[['case_id', 'date_decision']], on='case_id')\n#         cb = cb.with_columns((pl.col(pl.Date) - pl.col('date_decision')).dt.total_days())\n#         cb = cb.drop('date_decision')\n        return cb\n#     def transform_person_1(self, df, person_1):\n#         person_1_feats_1 = person_1.group_by(\"case_id\").agg(\n#             pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n#             (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n#         )\n#         person_1_feats_2 = person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n#             pl.col(\"num_group1\") == 0\n#         ).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n#         df = df.join(person_1_feats_1, on='case_id', how='left')\n#         df = df.join(person_1_feats_2, on='case_id', how='left')\n#         df = df.with_columns(pl.col('mainoccupationinc_384A_any_selfemployed').fill_null(False))\n#         return df\n#     def transform_cb_b_2(self, df, cb_b_2):\n#         cb_b_2_feats = cb_b_2.group_by(\"case_id\").agg(\n#             pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n#             (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n#         )\n#         df = df.join(cb_b_2_feats, on='case_id', how='left')\n#         return df\n    \n    def transform_depth_2(self, df):\n        cb_b_2 = read_file(os.path.join(self.path, f\"{self.mode}_credit_bureau_b_2.csv\"), depth=2)\n        df = df.join(cb_b_2, how='left', on='case_id', suffix='_cb_b_2')\n        del cb_b_2\n        return df \n        \n    def transform_depth_1(self, df):\n        app_prev = read_files(os.path.join(self.path, f\"{self.mode}_applprev_1_*.csv\"), depth=1)\n        df = df.join(app_prev, how='left', on='case_id', suffix='_app')\n        del app_prev\n        tax_a = read_file(os.path.join(self.path, f\"{self.mode}_tax_registry_a_1.csv\"), depth=1)\n        df = df.join(tax_a, how='left', on='case_id', suffix='_tax_a')\n        del tax_a\n        tax_b = read_file(os.path.join(self.path, f\"{self.mode}_tax_registry_b_1.csv\"), depth=1)\n        df = df.join(tax_b, how='left', on='case_id', suffix='_tax_b')\n        del tax_b\n        tax_c = read_file(os.path.join(self.path, f\"{self.mode}_tax_registry_c_1.csv\"), depth=1)\n        df = df.join(tax_c, how='left', on='case_id', suffix='_tax_c')\n        del tax_c\n        cb_a = read_files(os.path.join(self.path, f\"{self.mode}_credit_bureau_a_1_*.csv\"), depth=1)\n        df = df.join(cb_a, how='left', on='case_id', suffix='_cb_a')\n        del cb_a\n        cb_b = read_file(os.path.join(self.path, f\"{self.mode}_credit_bureau_b_1.csv\"), depth=1)\n        df = df.join(cb_b, how='left', on='case_id', suffix='_cb_b')\n        del cb_b\n        other = read_file(os.path.join(self.path, f\"{self.mode}_other_1.csv\"), depth=1)\n        df = df.join(other, how='left', on='case_id', suffix='_other')\n        del other\n        person = read_file(os.path.join(self.path, f\"{self.mode}_person_1.csv\"), depth=1)\n        df = df.join(person, how='left', on='case_id', suffix='_person')\n        del person                   \n        deposit = read_file(os.path.join(self.path, f\"{self.mode}_deposit_1.csv\"), depth=1)\n        df = df.join(deposit, how='left', on='case_id', suffix='_deposit')\n        del deposit\n        debit = read_file(os.path.join(self.path, f\"{self.mode}_debitcard_1.csv\"), depth=1)\n        df = df.join(debit, how='left', on='case_id', suffix='_debit')\n        del debit\n        return df \n\n    def transform_depth_0(self, base):\n        # static data transformation \n        static = read_files(os.path.join(self.path, f\"{self.mode}_static_0_*.csv\"))\n        static = self.transform_static(static)\n        # cb - credit bureau\n        cb = read_file(os.path.join(self.path, f\"{self.mode}_static_cb_0.csv\"))\n        cb = self.transform_cb(cb)\n        base = base.join(static, on='case_id', how='left').join(cb, on='case_id', how='left')\n        del static \n        del cb \n        return base\n    \n    def transform(self):\n        if self.mode == 'train':\n            base = read_file(Config.train_base_path)\n        else:\n            base = read_file(os.path.join(self.path, f\"{self.mode}_base.csv\"))\n        base = self.transform_base(base)\n        df = self.transform_depth_0(base)\n        df = self.transform_depth_1(df)\n        df = self.transform_depth_2(df)\n        df = df.pipe(Utilities.filter_cols).pipe(Utilities.handle_dates)\n        df = df.to_pandas()\n        exp_num = Config.experiment_number\n        version = Config.version \n        if self.mode == 'train':\n            cols_to_ignore = [f'fold_{i}' for i in range(Config.num_folds)]\n            cols_to_ignore.extend(['case_id', 'date_decision', 'MONTH', 'WEEK_NUM', 'target'])\n            selected_columns = [col for col in df.columns if col not in cols_to_ignore]\n            with open(f'./selected_columns_E{exp_num}_V{version}.pkl', 'wb') as f:\n                pickle.dump(selected_columns, f)\n            selected_cat_columns = []\n            for col in df.columns:\n                if col not in cols_to_ignore and df[col].dtype.name in ['object', 'string']:\n                    selected_cat_columns.append(col)\n            ordinal_encoder = OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1)\n            df.loc[:, selected_cat_columns] = ordinal_encoder.fit_transform(df.loc[:, selected_cat_columns])\n            with open(f'./selected_cat_columns_E{exp_num}_V{version}.pkl', 'wb') as f:\n                pickle.dump(selected_cat_columns, f)\n            with open(f'./ordinal_enc_E{exp_num}_V{version}.pkl', 'wb') as f:\n                pickle.dump(ordinal_encoder, f)\n        else:\n            model_path = Config.model_path\n            with open(f'{model_path}/selected_columns_E{exp_num}_V{version}.pkl', 'rb') as f:\n                selected_columns = pickle.load(f)\n            with open(f'{model_path}/selected_cat_columns_E{exp_num}_V{version}.pkl', 'rb') as f:\n                selected_cat_columns = pickle.load(f)\n            with open(f'{model_path}/ordinal_enc_E{exp_num}_V{version}.pkl', 'rb') as f:\n                ordinal_encoder = pickle.load(f)\n            df.loc[:, selected_cat_columns] = ordinal_encoder.transform(df.loc[:, selected_cat_columns])\n        df[selected_cat_columns] = df[selected_cat_columns].astype(float)\n        self.selected_columns = selected_columns\n        self.selected_cat_columns = selected_cat_columns\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:39:26.291199Z","iopub.execute_input":"2024-04-24T11:39:26.292061Z","iopub.status.idle":"2024-04-24T11:39:26.323718Z","shell.execute_reply.started":"2024-04-24T11:39:26.292030Z","shell.execute_reply":"2024-04-24T11:39:26.322844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformer = DataTransformer()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:39:26.447752Z","iopub.execute_input":"2024-04-24T11:39:26.448372Z","iopub.status.idle":"2024-04-24T11:39:26.452264Z","shell.execute_reply.started":"2024-04-24T11:39:26.448346Z","shell.execute_reply":"2024-04-24T11:39:26.451276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = transformer.transform()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:39:27.363183Z","iopub.execute_input":"2024-04-24T11:39:27.363911Z","iopub.status.idle":"2024-04-24T11:42:06.592450Z","shell.execute_reply.started":"2024-04-24T11:39:27.363880Z","shell.execute_reply":"2024-04-24T11:42:06.591611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Utilities.CleanMemory()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:42:06.594102Z","iopub.execute_input":"2024-04-24T11:42:06.594410Z","iopub.status.idle":"2024-04-24T11:42:06.959220Z","shell.execute_reply.started":"2024-04-24T11:42:06.594385Z","shell.execute_reply":"2024-04-24T11:42:06.958372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = reduce_mem_usage(data)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:42:06.960248Z","iopub.execute_input":"2024-04-24T11:42:06.960535Z","iopub.status.idle":"2024-04-24T11:42:13.512714Z","shell.execute_reply.started":"2024-04-24T11:42:06.960511Z","shell.execute_reply":"2024-04-24T11:42:13.511704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:42:13.514893Z","iopub.execute_input":"2024-04-24T11:42:13.515224Z","iopub.status.idle":"2024-04-24T11:42:13.976525Z","shell.execute_reply.started":"2024-04-24T11:42:13.515194Z","shell.execute_reply":"2024-04-24T11:42:13.975570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nums=[c for c in transformer.selected_columns if c not in transformer.selected_cat_columns]\nfrom itertools import combinations, permutations\n#df_train=df_train[nums]\nnans_df = data[nums].isna()\nnans_groups={}\nfor col in nums:\n    cur_group = nans_df[col].sum()\n    try:\n        nans_groups[cur_group].append(col)\n    except:\n        nans_groups[cur_group]=[col]\ndel nans_df; x=collect()\n\ndef reduce_group(grps):\n    use = []\n    for g in grps:\n        mx = 0; vx = g[0]\n        for gg in g:\n            n = data[gg].nunique()\n            if n>mx:\n                mx = n\n                vx = gg\n            #print(str(gg)+'-'+str(n),', ',end='')\n        use.append(vx)\n        #print()\n    print('Use these',use)\n    return use\n\ndef group_columns_by_correlation(matrix, threshold=0.8):\n    # 计算列之间的相关性\n    correlation_matrix = matrix.corr()\n\n    # 分组列\n    groups = []\n    remaining_cols = list(matrix.columns)\n    while remaining_cols:\n        col = remaining_cols.pop(0)\n        group = [col]\n        correlated_cols = [col]\n        for c in remaining_cols:\n            if correlation_matrix.loc[col, c] >= threshold:\n                group.append(c)\n                correlated_cols.append(c)\n        groups.append(group)\n        remaining_cols = [c for c in remaining_cols if c not in correlated_cols]\n    \n    return groups\n\nuses=[]\nfor k,v in nans_groups.items():\n    if len(v)>1:\n            Vs = nans_groups[k]\n            #cross_features=list(combinations(Vs, 2))\n            #make_corr(Vs)\n            grps= group_columns_by_correlation(data[Vs], threshold=0.8)\n            use=reduce_group(grps)\n            uses=uses+use\n            #make_corr(use)\n    else:\n        uses=uses+v\n    print('####### NAN count =',k)\nprint(uses)\nprint(len(uses))\ncols_to_ignore = [f'fold_{i}' for i in range(Config.num_folds)]\ncols_to_ignore.extend(['case_id', 'WEEK_NUM', 'target'])\nuses=uses+cols_to_ignore+transformer.selected_cat_columns\nprint(len(uses))\ndata=data[uses]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:44:03.115373Z","iopub.execute_input":"2024-04-24T11:44:03.116296Z","iopub.status.idle":"2024-04-24T11:44:21.398915Z","shell.execute_reply.started":"2024-04-24T11:44:03.116264Z","shell.execute_reply":"2024-04-24T11:44:21.397853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:44:30.631364Z","iopub.execute_input":"2024-04-24T11:44:30.631789Z","iopub.status.idle":"2024-04-24T11:44:30.639495Z","shell.execute_reply.started":"2024-04-24T11:44:30.631752Z","shell.execute_reply":"2024-04-24T11:44:30.638345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_df(df):\n    if Config.mode == 'train':\n        base = df[['case_id', 'WEEK_NUM', 'target']]\n        X = df[[c for c in df.columns if c not in ['case_id', 'WEEK_NUM', 'target']]]\n        y = df['target']\n        return base, X, y\n    else:\n        base = df[['case_id']]\n        X = df[transformer.selected_columns]\n        return base, X","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:47:30.847597Z","iopub.execute_input":"2024-04-24T11:47:30.848504Z","iopub.status.idle":"2024-04-24T11:47:30.856159Z","shell.execute_reply.started":"2024-04-24T11:47:30.848459Z","shell.execute_reply":"2024-04-24T11:47:30.854907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Utilities.CleanMemory()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:47:31.004114Z","iopub.execute_input":"2024-04-24T11:47:31.004752Z","iopub.status.idle":"2024-04-24T11:47:31.483872Z","shell.execute_reply.started":"2024-04-24T11:47:31.004713Z","shell.execute_reply":"2024-04-24T11:47:31.482857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"auc_train, auc_valid, auc_test = [], [], []\nstab_train, stab_valid, stab_test = [], [], []\nfor j in range(5):\n    col_name = f'fold_{j}'\n    train_data, test_data = data[data[col_name] == 1], data[data[col_name] == 0]\n    train_data, valid_data = train_test_split(train_data, train_size=0.8, random_state=123)\n    base_train, X_train, y_train = split_df(train_data)\n    base_valid, X_valid, y_valid = split_df(valid_data)\n    base_test, X_test, y_test = split_df(test_data)\n    lgb_train = lgb.Dataset(X_train, label=y_train)\n    lgb_valid = lgb.Dataset(X_valid, label=y_valid, reference=lgb_train)\n    params = {\n        \"boosting_type\": \"gbdt\",\n        \"objective\": \"binary\",\n        \"metric\": \"auc\",\n        \"max_depth\": 3,\n        \"num_leaves\": 31,\n        \"learning_rate\": 0.05,\n        \"feature_fraction\": 0.6,\n        \"bagging_fraction\": 0.8,\n        \"bagging_freq\": 5,\n        \"n_estimators\": 1000,\n        \"verbose\": -1,\n        \"device\": \"gpu\"\n    }\n\n    gbm = lgb.train(\n        params,\n        lgb_train,\n        valid_sets=lgb_valid,\n        callbacks=[lgb.log_evaluation(50), lgb.early_stopping(10)]\n    )\n    exp_num = Config.experiment_number\n    version = Config.version \n    with open(f'./lgb_model_E{exp_num}_V{version}_{j+1}.pkl', 'wb') as f:\n        pickle.dump(gbm, f)\n    for i, data_tup in enumerate([(base_train[['target', \"WEEK_NUM\"]], X_train), (base_valid[['target', \"WEEK_NUM\"]], X_valid), (base_test[['target', \"WEEK_NUM\"]], X_test)]):\n        base, X = data_tup\n        y_pred = gbm.predict(X, num_iteration=gbm.best_iteration)\n        base['score'] = y_pred\n        auc = roc_auc_score(base['target'], y_pred)\n        stability = gini_stability(base)\n        if i == 0:\n            print(f'The AUC score on the train set is: {auc}') \n            auc_train.append(auc)\n            print(f'The stability score on the train set is: {stability}') \n            stab_train.append(stability)\n        elif i == 1:\n            print(f'The AUC score on the valid set is: {auc}') \n            auc_valid.append(auc)\n            print(f'The stability score on the valid set is: {stability}') \n            stab_valid.append(stability)\n        else:\n            print(f'The AUC score on the test set is: {auc}') \n            auc_test.append(auc)\n            print(f'The stability score on the test set is: {stability}') \n            stab_test.append(stability)\n#     stab_train.append(gini_stability(base_train))\n#     stab_valid.append(gini_stability(base_valid))\n#     stab_test.append(gini_stability(base_test))","metadata":{"execution":{"iopub.status.busy":"2024-04-24T11:47:32.265207Z","iopub.execute_input":"2024-04-24T11:47:32.265686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'The AUC score on the train set is: {np.mean(auc_train)} with std {np.std(auc_train)}') \nprint(f'The AUC score on the valid set is: {np.mean(auc_valid)} with std {np.std(auc_valid)}') \nprint(f'The AUC score on the test set is: {np.mean(auc_test)} with std {np.std(auc_test)}') \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'The stability score on the train set is: {np.mean(stab_train)} with std {np.std(stab_train)}') \nprint(f'The stability score on the valid set is: {np.mean(stab_valid)} with std {np.std(stab_valid)}') \nprint(f'The stability score on the test set is: {np.mean(stab_test)} with std {np.std(stab_test)}') \n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T10:46:25.464278Z","iopub.execute_input":"2024-04-24T10:46:25.464671Z","iopub.status.idle":"2024-04-24T10:46:25.474166Z","shell.execute_reply.started":"2024-04-24T10:46:25.464642Z","shell.execute_reply":"2024-04-24T10:46:25.473286Z"},"trusted":true},"execution_count":null,"outputs":[]}]}