{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30716,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Import Libraries\n","metadata":{}},{"cell_type":"code","source":"!pip install kds","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:13:28.258676Z","iopub.execute_input":"2024-05-30T23:13:28.259409Z","iopub.status.idle":"2024-05-30T23:13:42.796979Z","shell.execute_reply.started":"2024-05-30T23:13:28.259380Z","shell.execute_reply":"2024-05-30T23:13:42.795911Z"},"_kg_hide-output":true,"_kg_hide-input":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nfrom typing import List, Tuple\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\nimport joblib\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\nimport functools\nfrom functools import partial\nimport gc \nimport time\nimport kds\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report, fbeta_score, roc_auc_score, roc_curve, precision_recall_curve\nfrom sklearn.feature_selection import SelectKBest\nfrom sklearn.feature_selection import chi2\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.feature_selection import RFE\nfrom sklearn.linear_model import LogisticRegression\nfrom lightgbm import LGBMClassifier\nfrom sklearn.feature_selection import SelectFromModel\nfrom sklearn.model_selection._split import _BaseKFold, indexable, _num_samples\nfrom sklearn.utils.validation import _deprecate_positional_args\nfrom hyperopt import fmin, hp, tpe, Trials, space_eval, STATUS_OK, STATUS_RUNNING\nimport warnings\nwarnings.filterwarnings('ignore')\n# display full dataframe\npd.set_option(\"display.max_rows\", None)\npd.set_option(\"display.max_columns\", None)\nwarnings.filterwarnings('ignore')\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-30T23:13:45.621959Z","iopub.execute_input":"2024-05-30T23:13:45.622326Z","iopub.status.idle":"2024-05-30T23:13:51.108718Z","shell.execute_reply.started":"2024-05-30T23:13:45.622295Z","shell.execute_reply":"2024-05-30T23:13:51.107942Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Helper Functions","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\n\n\nclass Pipeline:\n\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.7:\n                    df = df.drop(col)\n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n        \n        return df\n    \nclass Aggregator:\n    # Please add or subtract features yourself, be aware that too many features will take up too much space.\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        # expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n#         expr_median = [pl.median(col).alias(f\"median_{col}\") for col in cols]\n        expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n\n        return expr_max + expr_last + expr_mean + expr_var\n#     + expr_median \n\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        # expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n#         expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n#         expr_median = [pl.median(col).alias(f\"median_{col}\") for col in cols]\n\n        return expr_max + expr_last + expr_first\n#     + expr_mean + expr_median\n\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        # expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        # expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n        return expr_max + expr_last + expr_count\n\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        # expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return expr_max + expr_last + expr_min\n\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        # expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        # expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return expr_max + expr_last\n\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs\n    \n    \ndef read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    if depth in [1,2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df)) \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols\n\n\n\ndef fill_na(df: pd.DataFrame, drop_percentage=None, fill_value=0, impute_strategy='mean') -> pd.DataFrame:\n    \"\"\"\n\n    Impute features having NA values.\n\n    Parameters:\n        df(pd.DataFrame): Dataframe to impute NA values\n        drop_percentage(float): percentage of NA values to drop rows.\n        fill_value(int or float): value to fill NA with.\n        impute_strategy(string): mean, median, or mode\n\n    Returns:\n        pd.DataFrame: Imputed Dataframe after NA treatment.\n    \"\"\"\n    df = df.copy()\n    # apply only for numeric columns\n    numeric_columns = df.select_dtypes([np.number]).columns\n    # TODO expand for categorical columns too\n    # if percentage of na is higher than defined threshold drop rows\n    if drop_percentage:\n        df = df.dropna(thresh=int(df.shape[0] * (1 - drop_percentage)), axis=0)\n    if fill_value < 0:\n        # fill numeric columns with the fill_value\n        df[numeric_columns] = df[numeric_columns].fillna(fill_value)\n    else:\n        if impute_strategy == 'mean':\n            imputer = SimpleImputer(missing_values=np.nan, strategy='mean')\n        elif impute_strategy == 'median':\n            imputer = SimpleImputer(missing_values=np.nan, strategy='median')\n        elif impute_strategy == 'mode':\n            imputer = SimpleImputer(\n                missing_values=np.nan, strategy='most_frequent')\n        # apply imputer to numeric columns\n        df[numeric_columns] = imputer.fit_transform(df[numeric_columns])\n\n    return df\n\ndef create_xy(df: pd.DataFrame, target: str) -> Tuple[pd.DataFrame, pd.Series]:\n    \"\"\"\n    Creates X and Y for a given Dataframe and target column.\n\n    Parameters:\n        df (pd.Dataframe): The input dataframe\n        target (str): The target column name\n\n    Returns:\n        Tuple[pd.Dataframe, pd.Series]: A tuple containing X and Y.\n            X : features dataframe\n            Y : target dataframe\n\n    Raises:\n        ValueError: if target column does not exist in input Dataframe\n    \"\"\"\n    # add input validation to check if column exists in the input dataframe\n    if target not in df.columns:\n        raise ValueError(\n            f\"Target column '{target}' not found in input DataFrame.\")\n    # set x and y\n    y = df.pop(target).values\n    # TODO use it later if necessary\n    target_distribution = sum(y) / len(y)\n    print(f\"Target Distribution is: {target_distribution}\")\n    return df, y\n\ndef df_set_indices(df: pd.DataFrame, id_level: str, date_col: str) -> pd.DataFrame:\n    df = df.copy()\n    df.set_index([date_col, id_level], inplace=True)\n    return df\n\n'''Create function for providing summary statistics in a table'''\ndef resumetable(df):\n    \"\"\"\n    Objective: For a given dataframe this function provides information\n    regarding Missing and Unique values per column.\n\n    Input: param df: Dataframe to check the information.\n\n    Output: return summary: a dataframe with columns providing summary per column of the input dataframe.\n    \n    \"\"\"\n    print(f\"Dataset Shape: {df.shape}\")\n    summary = pd.DataFrame(df.dtypes,columns=['dtypes'])\n    summary = summary.reset_index()\n    summary['Name'] = summary['index']\n    summary = summary[['Name','dtypes']]\n    summary['Missing'] = df.isna().sum().values\n    summary['Missing Percentage'] = df.isna().sum().values/len(df)\n    summary['Uniques'] = df.nunique().values\n    return summary\n\nclass MissingValuesInsights:\n    @classmethod\n    def insights(cls, df, perc_limit):\n        #Create missing value df\n        missing = cls.count_sort(df)\n        #Plot missing values\n        cls.plot(df,perc_limit)\n        return missing\n    '''Create function for plotting missing features'''\n    @classmethod\n    def plot(cls, df, perc_limit):\n        total = df.isnull().sum().sort_values(ascending=False)\n        percent = (df.isnull().sum()/df.isnull().count()).sort_values(ascending=False)\n        missing_data = pd.concat([total, percent], axis=1,join='outer', keys=['missing_count', 'percent'])\n        missing_data = missing_data.loc[missing_data['percent'] > perc_limit]\n        missing_data = missing_data.sort_values(by='missing_count')\n        missing_data = missing_data[['missing_count']].drop_duplicates()\n        #\n        ind = np.arange(missing_data.shape[0])\n        width = 0.2\n        fig, ax = plt.subplots(figsize=(15,5))\n        rects = ax.barh(ind, missing_data.missing_count.values, color='b')\n        ax.set_yticks(ind)\n        ax.set_yticklabels(missing_data.index.values, rotation='horizontal')\n        ax.set_xlabel(\"Missing Observations Count\")\n        ax.set_title(\"Missing Observations Count - Features\")\n        plt.show()\n    @classmethod\n    def count_sort(cls, df):\n        total = df.isnull().sum().sort_values(ascending=False)\n        percent = (df.isnull().sum()/df.isnull().count()).sort_values(ascending=False)\n        missing_data = pd.concat([total, percent], axis=1,join='outer', keys=['Total Missing Count', '% of Total Observations'])\n        missing_data = missing_data.loc[missing_data['% of Total Observations']>=0.05]\n        missing_data.index.name =' Feature'\n        missing_data = missing_data.reset_index()\n        return missing_data\n\n    \n    \nclass FeatureSelection:\n\n    @classmethod\n    def cor_selector(cls, X, y, features_to_select, feature_name):\n        cor_list = []\n        # calculate the correlation with y for each feature\n        for i in X.columns.tolist():\n            cor = np.corrcoef(X[i], y)[0, 1]\n            cor_list.append(cor)\n        # replace NaN with 0\n        cor_list = [0 if np.isnan(i) else i for i in cor_list]\n        # feature name\n        cor_feature = X.iloc[:, np.argsort(\n            np.abs(cor_list))[-features_to_select:]].columns.tolist()\n        # feature selection? 0 for not select, 1 for select\n        cor_support = [\n            True if i in cor_feature else False for i in feature_name]\n        return cor_support, cor_feature\n\n    @classmethod\n    def remove_collinear_features(cls, x, threshold):\n        '''\n        Objective:\n            Remove collinear features in a dataframe with a correlation coefficient\n            greater than the threshold. Removing collinear features can help a model \n            to generalize and improves the interpretability of the model.\n\n        Inputs: \n            x: features dataframe\n            threshold: features with correlations greater than this value are removed\n\n        Output: \n            dataframe that contains only the non-highly-collinear features\n        '''\n\n        # Calculate the correlation matrix\n        corr_matrix = x.corr()\n        iters = range(len(corr_matrix.columns) - 1)\n        drop_cols = []\n\n        # Iterate through the correlation matrix and compare correlations\n        for i in iters:\n            for j in range(i+1):\n                item = corr_matrix.iloc[j:(j+1), (i+1):(i+2)]\n                col = item.columns\n                row = item.index\n                val = abs(item.values)\n\n                # If correlation exceeds the threshold\n                if val >= threshold:\n                    # Print the correlated features and the correlation value\n                    # print(col.values[0], \"|\", row.values[0], \"|\", round(val[0][0], 2))\n                    drop_cols.append(col.values[0])\n\n        # Drop one of each pair of correlated columns\n        drops = set(drop_cols)\n        x = x.drop(columns=drops)\n        print('Removed Columns {}'.format(drops))\n        return x\n\n    # dataframe from upper step and number of best variables to output\n\n    @classmethod\n    def variable_selection(cls,\n                           X,\n                           y,\n                           best,\n                           features_to_select):\n\n        start = time.time()\n\n # =============================================================================\n #        #Remove missing\n # =============================================================================\n        print('Remove Missing..')\n        for col in X.columns:\n            if X[col].isnull().sum()/X.shape[0] > 0.9:\n                print(col)\n                del X[col]\n # =============================================================================\n #        #Remove low variance features\n # =============================================================================\n        for col in X.columns:\n            if X[col].nunique() == 1:\n                print(col)\n                del X[col]\n # =============================================================================\n #        #Fill na with -1\n # =============================================================================\n        X = fill_na(X, fill_value=-1)\n # =============================================================================\n #        #Pearson Correlation\n # =============================================================================\n        print('Pearson Correlation..')\n\n        feature_name = X.columns.tolist()\n\n        cor_support, cor_feature = cls.cor_selector(\n            X, y, features_to_select, feature_name)\n\n # =============================================================================\n #        #Chi-2\n # =============================================================================\n        print('Chi-2..')\n\n        X_norm = MinMaxScaler().fit_transform(X)\n        chi_selector = SelectKBest(chi2, k=features_to_select)\n        chi_selector.fit(X_norm, y)\n\n        chi_support = chi_selector.get_support()\n        chi_feature = X.loc[:, chi_support].columns.tolist()\n        print(str(len(chi_feature)), 'selected features')\n\n # =============================================================================\n #        #Wrapper RFE selector\n # =============================================================================\n#         print('RFE selector..')\n#         rfe_selector = RFE(estimator=LogisticRegression(\n#             class_weight='balanced'), n_features_to_select=features_to_select, step=150, verbose=5)\n#         rfe_selector.fit(X_norm, y)\n\n#         rfe_support = rfe_selector.get_support()\n#         rfe_feature = X.loc[:, rfe_support].columns.tolist()\n#         print(str(len(rfe_feature)), 'selected features')\n\n # =============================================================================\n #        #Embedded\n # =============================================================================\n\n#         print('LogisticRegression l2 ..')\n\n#         embeded_lr_selector = SelectFromModel(\n#             LogisticRegression(penalty=\"l2\", class_weight='balanced'))\n#         embeded_lr_selector.fit(X_norm, y)\n\n#         embeded_lr_support = embeded_lr_selector.get_support()\n#         embeded_lr_feature = X.loc[:, embeded_lr_support].columns.tolist()\n#         print(str(len(embeded_lr_feature)), 'selected features')\n # =============================================================================\n #        #Random Forest\n # =============================================================================\n        print('Random Forest ..')\n\n        embeded_rf_selector = SelectFromModel(RandomForestClassifier(\n            n_estimators=features_to_select, class_weight='balanced'))\n        embeded_rf_selector.fit(X, y)\n\n        embeded_rf_support = embeded_rf_selector.get_support()\n        embeded_rf_feature = X.loc[:, embeded_rf_support].columns.tolist()\n        print(str(len(embeded_rf_feature)), 'selected features')\n\n # =============================================================================\n #        #Light GBM\n # =============================================================================\n        print('Light GBM ..')\n\n        lgbc = LGBMClassifier(n_estimators=features_to_select,\n                              class_weight='balanced', learning_rate=0.05)\n\n        embeded_lgb_selector = SelectFromModel(lgbc)\n        embeded_lgb_selector.fit(X, y)\n\n        embeded_lgb_support = embeded_lgb_selector.get_support()\n        embeded_lgb_feature = X.loc[:, embeded_lgb_support].columns.tolist()\n        print(str(len(embeded_lgb_feature)), 'selected features')\n\n        pd.set_option('display.max_rows', None)\n        # put all selection together\n        feature_selection_df = pd.DataFrame({'Feature': feature_name,\n                                             'Pearson': cor_support,\n                                             'Chi-2': chi_support,\n#                                              'RFE': rfe_support,\n#                                              'Logistics': embeded_lr_support,\n                                             'Random Forest': embeded_rf_support,\n                                             'LightGBM': embeded_lgb_support\n                                             })\n\n        # count the selected times for each feature\n        # numeric_columns = feature_selection_df.select_dtypes(include=np.number)\n        # feature_selection_df[numeric_columns.columns] = numeric_columns.astype(int)\n        feature_selection_df['Pearson'] = np.where(((feature_selection_df['Pearson'] == False)),\n                                                   0,\n                                                   1)\n        feature_selection_df['Chi-2'] = np.where(((feature_selection_df['Chi-2'] == False)),\n                                                 0,\n                                                 1)\n#         feature_selection_df['RFE'] = np.where(((feature_selection_df['RFE'] == False)),\n#                                                0,\n#                                                1)\n#         feature_selection_df['Logistics'] = np.where(((feature_selection_df['Logistics'] == False)),\n#                                                      0,\n#                                                      1)\n        feature_selection_df['Random Forest'] = np.where(((feature_selection_df['Random Forest'] == False)),\n                                                         0,\n                                                         1)\n        feature_selection_df['LightGBM'] = np.where(((feature_selection_df['LightGBM'] == False)),\n                                                    0,\n                                                    1)\n        print(feature_selection_df.head(4))\n        feature_selection_df.set_index('Feature', inplace=True)\n        feature_selection_df['Total'] = np.sum(feature_selection_df, axis=1)\n        # display the top x\n        feature_selection_df.reset_index(inplace=True)\n        feature_selection_df = feature_selection_df.sort_values(\n            ['Total', 'Feature'], ascending=False)\n        feature_selection_df.index = range(1, len(feature_selection_df)+1)\n\n        best_features = feature_selection_df.head(best)\n        best_features = list(best_features['Feature'])\n        print(best_features)\n        Xbest = X[best_features]\n\n        # remove uncorrelated features\n        Xbest = cls.remove_collinear_features(Xbest, 0.9)\n        # keep those features only\n        feat_list = list(Xbest.columns)\n        print(\"Selected Features are: \")\n        # print chosen vars\n        print(feat_list)\n\n        end = time.time()\n        print((end - start)/60)\n\n        return feat_list   \n\n    \nclass LabelEncodingColumns(BaseEstimator, TransformerMixin):\n    def __init__(self, cols=None):\n        self.cols = cols\n        self.label_encoders = {col: LabelEncoder() for col in cols}\n        self.feature_dictionary = {}\n        self.final_dictionary = {}\n\n    def fit(self, df, y=None):\n        for col in self.cols:\n            self.label_encoders[col].fit(\n                pd.concat([df[col], pd.Series('unknown')]))\n            self.feature_dictionary[col] = self.label_encoders[col].classes_\n\n            le_name_mapping = dict(\n                zip(\n                    self.label_encoders[col].classes_,\n                    self.label_encoders[col].transform(\n                        self.label_encoders[col].classes_\n                    ),\n                )\n            )\n            self.final_dictionary.update({col: le_name_mapping})\n\n        return self\n\n    def transform(self, df_, y=None):\n\n        df = df_.copy()\n\n        label_enc_dict = {}\n\n        for col in self.cols:\n            temp_dic = dict(\n                zip(\n                    self.label_encoders[col].classes_, self.label_encoders[col].classes_\n                )\n            )\n            df[col] = df[col].map(temp_dic)\n            df[col].fillna('unknown', inplace=True)\n            label_enc_dict[col] = self.label_encoders[col].transform(df[col])\n            del temp_dic\n\n        # The index of the resulting DataFrame should be assigned and\n        # equal to the one of the original DataFrame. Otherwise, upon\n        # concatenation NaNs will be introduced.\n        labelenc_cols = pd.DataFrame(label_enc_dict, index=df.index)\n\n        for col in self.cols:\n            df[col] = labelenc_cols[col]\n\n        return df\n\nclass GroupTimeSeriesSplit(_BaseKFold):\n    \"\"\"Time Series cross-validator variant with non-overlapping groups.\n    Provides train/test indices to split time series data samples\n    that are observed at fixed time intervals according to a\n    third-party provided group.\n    In each split, test indices must be higher than before, and thus shuffling\n    in cross validator is inappropriate.\n    This cross-validation object is a variation of :class:`KFold`.\n    In the kth split, it returns first k folds as train set and the\n    (k+1)th fold as test set.\n    The same group will not appear in two different folds (the number of\n    distinct groups has to be at least equal to the number of folds).\n    Note that unlike standard cross-validation methods, successive\n    training sets are supersets of those that come before them.\n    Read more in the :ref:`User Guide <cross_validation>`.\n    Parameters\n    ----------\n    n_splits : int, default=5\n        Number of splits. Must be at least 2.\n    max_train_size : int, default=None\n        Maximum size for a single training set.\n    Examples\n    --------\n    >>> import numpy as np\n    >>> from sklearn.model_selection import GroupTimeSeriesSplit\n    >>> groups = np.array(['a', 'a', 'a', 'a', 'a', 'a',\\\n                           'b', 'b', 'b', 'b', 'b',\\\n                           'c', 'c', 'c', 'c',\\\n                           'd', 'd', 'd'])\n    >>> gtss = GroupTimeSeriesSplit(n_splits=3)\n    >>> for train_idx, test_idx in gtss.split(groups, groups=groups):\n    ...     print(\"TRAIN:\", train_idx, \"TEST:\", test_idx)\n    ...     print(\"TRAIN GROUP:\", groups[train_idx],\\\n                  \"TEST GROUP:\", groups[test_idx])\n    TRAIN: [0, 1, 2, 3, 4, 5] TEST: [6, 7, 8, 9, 10]\n    TRAIN GROUP: ['a' 'a' 'a' 'a' 'a' 'a']\\\n    TEST GROUP: ['b' 'b' 'b' 'b' 'b']\n    TRAIN: [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10] TEST: [11, 12, 13, 14]\n    TRAIN GROUP: ['a' 'a' 'a' 'a' 'a' 'a' 'b' 'b' 'b' 'b' 'b']\\\n    TEST GROUP: ['c' 'c' 'c' 'c']\n    TRAIN: [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14]\\\n    TEST: [15, 16, 17]\n    TRAIN GROUP: ['a' 'a' 'a' 'a' 'a' 'a' 'b' 'b' 'b' 'b' 'b' 'c' 'c' 'c' 'c']\\\n    TEST GROUP: ['d' 'd' 'd']\n    \"\"\"\n    @_deprecate_positional_args\n    def __init__(self,\n                 n_splits=5,\n                 *,\n                 max_train_size=None\n                 ):\n        super().__init__(n_splits, shuffle=False, random_state=None)\n        self.max_train_size = max_train_size\n\n    def split(self, X, y=None, groups=None):\n        \"\"\"Generate indices to split data into training and test set.\n        Parameters\n        ----------\n        X : array-like of shape (n_samples, n_features)\n            Training data, where n_samples is the number of samples\n            and n_features is the number of features.\n        y : array-like of shape (n_samples,)\n            Always ignored, exists for compatibility.\n        groups : array-like of shape (n_samples,)\n            Group labels for the samples used while splitting the dataset into\n            train/test set.\n        Yields\n        ------\n        train : ndarray\n            The training set indices for that split.\n        test : ndarray\n            The testing set indices for that split.\n        \"\"\"\n        if groups is None:\n            raise ValueError(\n                \"The 'groups' parameter should not be None\")\n        X, y, groups = indexable(X, y, groups)\n        n_samples = _num_samples(X)\n        n_splits = self.n_splits\n        n_folds = n_splits + 1\n        group_dict = {}\n        u, ind = np.unique(groups, return_index=True)\n        unique_groups = u[np.argsort(ind)]\n        n_samples = _num_samples(X)\n        n_groups = _num_samples(unique_groups)\n        for idx in np.arange(n_samples):\n            if (groups[idx] in group_dict):\n                group_dict[groups[idx]].append(idx)\n            else:\n                group_dict[groups[idx]] = [idx]\n        if n_folds > n_groups:\n            raise ValueError(\n                (\"Cannot have number of folds={0} greater than\"\n                 \" the number of groups={1}\").format(n_folds,\n                                                     n_groups))\n        group_test_size = n_groups // n_folds\n        group_test_starts = range(n_groups - n_splits * group_test_size,\n                                  n_groups, group_test_size)\n        for group_test_start in group_test_starts:\n            train_array = []\n            test_array = []\n            for train_group_idx in unique_groups[:group_test_start]:\n                train_array_tmp = group_dict[train_group_idx]\n                train_array = np.sort(np.unique(\n                                      np.concatenate((train_array,\n                                                      train_array_tmp)),\n                                      axis=None), axis=None)\n            train_end = train_array.size\n            if self.max_train_size and self.max_train_size < train_end:\n                train_array = train_array[train_end -\n                                          self.max_train_size:train_end]\n            for test_group_idx in unique_groups[group_test_start:\n                                                group_test_start +\n                                                group_test_size]:\n                test_array_tmp = group_dict[test_group_idx]\n                test_array = np.sort(np.unique(\n                    np.concatenate((test_array,\n                                    test_array_tmp)),\n                    axis=None), axis=None)\n            yield [int(i) for i in train_array], [int(i) for i in test_array]\n\nclass OptimizationFunctions:\n\n    '''Function for lgbm evaluation metric similar to roc auc'''\n    @classmethod\n    def gini(cls, y, pred):\n        g = np.asarray(np.c_[y, pred, np.arange(len(y))], dtype=float)\n        g = g[np.lexsort((g[:, 2], -1*g[:, 1]))]\n        gs = g[:, 0].cumsum().sum() / g[:, 0].sum()\n        gs -= (len(y) + 1) / 2.\n        return gs / len(y)\n\n    '''Function for lgbm evaluation metric'''\n    @classmethod\n    def gini_lgb(cls, y_hat, data):\n        y = list(data.get_label())\n        score = cls.gini(y, y_hat) / cls.gini(y, y)\n        return 'gini', score, True\n\n    '''Function for lgbm evaluation metric'''\n    def lgb_fbeta_score(cls, y_hat, data):\n        y_true = data.get_label()\n        y_hat = np.round(y_hat)  # scikits f1 doesn't like probabilities\n        return 'fbeta', fbeta_score(y_true, y_hat, average='binary', beta=0.5), True\n\n    '''Function for lgbm evaluation metric'''\n    def lgb_precision_score(cls, y_hat, data):\n        y_true = data.get_label()\n        y_hat = np.round(y_hat)\n        return 'precision', precision_score(y_true, y_hat), True\n\n\nclass LightgbmPipeline:\n\n    @classmethod\n    def pipeline(cls,\n                 x_train,\n                 y_train,\n                 hyper_space,\n                 scoring,\n                 time_train,\n                 number_of_evals):\n\n        # run optimization\n        optimization = fmin(fn=partial(cls.time_series_to_minimize,\n                                       scoring=scoring,\n                                       time_train=time_train,\n                                       x_train=x_train,\n                                       y_train=y_train),\n                            space=hyper_space,\n                            algo=tpe.suggest,\n                            trials=Trials(),\n                            max_evals=number_of_evals)\n\n        # fit model\n        best_params = space_eval(hyper_space, optimization)\n        #\n        lgbm_chosen = cls.run_lgb(\n            x_train, y_train, best_params)\n\n        return lgbm_chosen, best_params\n\n    @classmethod\n    def run_lgb(cls, x_train, y_train, best_params):\n        # Init params for lgbm\n        params_lgbm = {'boosting_type': 'gbdt',\n                       'max_depth': 8,\n                       'objective': 'binary',\n#                        'nthread': 5,\n                       \"device\": \"gpu\",\n                       'num_leaves': 120,\n                       'learning_rate': 0.15,\n                       'max_bin': 512,\n                       'subsample_for_bin': 100,\n                       'subsample': 0.65,\n                       'subsample_freq': 1,\n                       'colsample_bytree': 0.65,\n                       'reg_alpha': 2,\n                       'reg_lambda': 2,\n                       'min_split_gain': 0.1,\n                       'min_child_weight': 1,\n                       'min_child_samples': 2,\n                       'scale_pos_weight': 10,\n                       'num_class': 1,\n                       'silent': True,\n                       'verbose': -1,\n                       'metric': {'auc'},\n                       'eval_metric': {'auc'}\n                       }\n\n        # Using parameters already set above, replace in the best from the grid search\n        params_lgbm['learning_rate'] = best_params['learning_rate']\n        params_lgbm['scale_pos_weight'] = best_params['scale_pos_weight']\n        params_lgbm['max_depth'] = best_params['max_depth']\n        params_lgbm['num_leaves'] = best_params['num_leaves']\n#         params_lgbm['n_estimators'] = best_params['n_estimators']\n        params_lgbm['min_child_samples'] = best_params['min_child_samples']\n        params_lgbm['reg_lambda'] = best_params['reg_lambda']\n        params_lgbm['reg_alpha'] = best_params['reg_alpha']\n        params_lgbm['colsample_bytree'] = best_params['colsample_bytree']\n        params_lgbm['subsample'] = best_params['subsample']\n\n        # define dataset\n        train_data_df = lgb.Dataset(x_train, label=y_train)\n        evals_results_df = {}\n        print(\"Training the model...\")\n\n        lgbm_best = lgb.train(params_lgbm,\n                              train_data_df,\n                              #   evals_result=evals_results_df,\n                              num_boost_round=best_params['n_estimators'],\n#                               callbacks=[lgb.early_stopping(\n#                                   stopping_rounds=50)],\n#                               feval=OptimizationFunctions.gini_lgb,\n                              #   verbose_eval=True\n                              )\n\n        return lgbm_best\n\n    @classmethod\n    def time_series_to_minimize(cls,\n                                hyperparameters,\n                                scoring,\n                                time_train,\n                                x_train,\n                                y_train):\n\n        folds = 5\n        res_vec = np.zeros((folds, 1))\n        prv = np.zeros((x_train.shape[0], folds))\n        for (ii, (id0, id1)) in enumerate(GroupTimeSeriesSplit(n_splits=folds).split(x_train, groups=pd.DataFrame(time_train)['order'])):\n\n            x0, x1 = x_train.iloc[id0], x_train.iloc[id1]\n            y0, y1 = y_train[id0], y_train[id1]\n\n            model = LGBMClassifier(**hyperparameters)\n\n            model.fit(x0, y0, eval_metric=scoring,\n                      eval_set=[(x0, y0), (x1, y1)])\n\n            val_preds = model.predict(x1)\n\n            # validation score\n            score = roc_auc_score(y1, val_preds)\n            print(\"validation score: \" + str(score))\n            res_vec[ii] = score\n\n            del model, x0, x1, y0, y1\n        return -res_vec.mean()\n    \nclass PlotMetrics:\n\n    @classmethod\n    def plot_confusion_matrix(cls, y_true, y_pred, normalize=False, title=None, cmap=plt.cm.Blues):\n        \"\"\"\n        This function prints and plots the confusion matrix.\n        Normalization can be applied by setting `normalize=True`.\n        \"\"\"\n        if not title:\n            if normalize:\n                title = 'Normalized confusion matrix'\n            else:\n                title = 'Confusion matrix, without normalization'\n\n        # Compute confusion matrix\n        cm = confusion_matrix(y_true, y_pred)\n\n        if normalize:\n            cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n            print(\"Normalized confusion matrix\")\n        else:\n            print('Confusion matrix, without normalization')\n\n        print(cm)\n\n        fig, ax = plt.subplots()\n        im = ax.imshow(cm, interpolation='nearest', cmap=cmap)\n        ax.figure.colorbar(im, ax=ax)\n        # We want to show all ticks...\n        ax.set(xticks=[],\n               yticks=[],\n               # ... and label them with the respective list entries\n               # xticklabels=classes, yticklabels=classes,\n               title=title,\n               ylabel='True label',\n               xlabel='Predicted label')\n\n        # Rotate the tick labels and set their alignment.\n        plt.setp(ax.get_xticklabels(), rotation=45, ha=\"right\",\n                 rotation_mode=\"anchor\")\n\n        # Loop over data dimensions and create text annotations.\n        fmt = '.2f' if normalize else 'd'\n        thresh = cm.max() / 2.\n        for i in range(cm.shape[0]):\n            for j in range(cm.shape[1]):\n                ax.text(j, i, format(cm[i, j], fmt),\n                        ha=\"center\", va=\"center\",\n                        color=\"white\" if cm[i, j] > thresh else \"black\")\n        fig.tight_layout()\n        return ax\n\n    @classmethod\n    def plot_precision_recall_vs_thresholds(cls, precisions, recalls, thresholds):\n        plt.plot(thresholds, precisions[:-1], \"b--\", label=\"Precision\")\n        plt.plot(thresholds, recalls[:-1], \"g--\", label=\"Recall\")\n        plt.xlabel(\"Threshold\")\n        plt.legend(bbox_to_anchor=(1.05, 1),\n                   loc='upper left', borderaxespad=0.)\n        plt.grid(which=\"both\", axis=\"both\", color='gray',\n                 linestyle='-', linewidth=1)\n        plt.title('Precision-Recall Thresholds')\n\n    @classmethod\n    def plot_predictions(cls, x, y, clf):\n        # Predict on test set\n        # Predicting proba\n        predictions_clf_prob = clf.predict(x)\n        # Turn probability to 0-1 binary output\n        predictions_clf_01 = np.where(predictions_clf_prob > 0.5, 1, 0)\n\n        print(\"accuracy is:\", accuracy_score(y, predictions_clf_01))\n        print(\"\\n\")\n        print(\"confusion matrix is:\", confusion_matrix(y, predictions_clf_01))\n        print(\"\\n\")\n        print(\"fbeta is:\", fbeta_score(y, predictions_clf_01, beta=2))\n        print(\"\\n\")\n        print(classification_report(y, predictions_clf_01))\n\n        # Generate ROC curve values: fpr, tpr, thresholds\n        fpr, tpr, thresholds = roc_curve(y, predictions_clf_prob)\n\n        # Plot ROC curve\n        plt.plot([0, 1], [0, 1], 'k--')\n        plt.plot(fpr, tpr)\n        plt.xlabel('False Positive Rate')\n        plt.ylabel('True Positive Rate')\n        plt.title('ROC Curve')\n        plt.show()\n\n        # Plot non-normalized confusion matrix\n        cls.plot_confusion_matrix(y, predictions_clf_01,\n                                  title='Confusion matrix, without normalization')\n\n        # Plot normalized confusion matrix\n        cls.plot_confusion_matrix(y, predictions_clf_01, normalize=True,\n                                  title='Normalized confusion matrix')\n        plt.show()\n        # Plot Precision Recall Curve\n        precisions, recalls, thresholds = precision_recall_curve(\n            y, predictions_clf_prob)\n        cls.plot_precision_recall_vs_thresholds(\n            precisions, recalls, thresholds)\n        plt.show()\n        # Report\n        kds.metrics.report(y, predictions_clf_prob)\n\n        return predictions_clf_prob, predictions_clf_01\n\n    @classmethod\n    def plot_importance(cls, model, X, num=10):\n        feature_imp = pd.DataFrame(\n            {'Value': model.feature_importance(), 'Feature': X.columns})\n        plt.figure(figsize=(20, 10))\n        sns.set(font_scale=1)\n        sns.barplot(x=\"Value\", y=\"Feature\", data=feature_imp.sort_values(by=\"Value\",\n                                                                         ascending=False)[0:num])\n        plt.title('LightGBM Features (avg over folds)')\n        plt.tight_layout()\n        plt.savefig('lgbm_importance.png')\n        plt.show()\n\n        \nclass PlotMetrics:\n\n    @classmethod\n    def plot_confusion_matrix(cls, y_true, y_pred, normalize=False, title=None, cmap=plt.cm.Blues):\n        \"\"\"\n        This function prints and plots the confusion matrix.\n        Normalization can be applied by setting `normalize=True`.\n        \"\"\"\n        if not title:\n            if normalize:\n                title = 'Normalized confusion matrix'\n            else:\n                title = 'Confusion matrix, without normalization'\n\n        # Compute confusion matrix\n        cm = confusion_matrix(y_true, y_pred)\n\n        if normalize:\n            cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n            print(\"Normalized confusion matrix\")\n        else:\n            print('Confusion matrix, without normalization')\n\n        print(cm)\n\n        fig, ax = plt.subplots()\n        im = ax.imshow(cm, interpolation='nearest', cmap=cmap)\n        ax.figure.colorbar(im, ax=ax)\n        # We want to show all ticks...\n        ax.set(xticks=[],\n               yticks=[],\n               # ... and label them with the respective list entries\n               # xticklabels=classes, yticklabels=classes,\n               title=title,\n               ylabel='True label',\n               xlabel='Predicted label')\n\n        # Rotate the tick labels and set their alignment.\n        plt.setp(ax.get_xticklabels(), rotation=45, ha=\"right\",\n                 rotation_mode=\"anchor\")\n\n        # Loop over data dimensions and create text annotations.\n        fmt = '.2f' if normalize else 'd'\n        thresh = cm.max() / 2.\n        for i in range(cm.shape[0]):\n            for j in range(cm.shape[1]):\n                ax.text(j, i, format(cm[i, j], fmt),\n                        ha=\"center\", va=\"center\",\n                        color=\"white\" if cm[i, j] > thresh else \"black\")\n        fig.tight_layout()\n        return ax\n\n    @classmethod\n    def plot_precision_recall_vs_thresholds(cls, precisions, recalls, thresholds):\n        plt.plot(thresholds, precisions[:-1], \"b--\", label=\"Precision\")\n        plt.plot(thresholds, recalls[:-1], \"g--\", label=\"Recall\")\n        plt.xlabel(\"Threshold\")\n        plt.legend(bbox_to_anchor=(1.05, 1),\n                   loc='upper left', borderaxespad=0.)\n        plt.grid(which=\"both\", axis=\"both\", color='gray',\n                 linestyle='-', linewidth=1)\n        plt.title('Precision-Recall Thresholds')\n\n    @classmethod\n    def plot_predictions(cls, x, y, clf):\n        # Predict on test set\n        # Predicting proba\n        predictions_clf_prob = clf.predict(x)\n        # Turn probability to 0-1 binary output\n        predictions_clf_01 = np.where(predictions_clf_prob > 0.5, 1, 0)\n\n        print(\"accuracy is:\", accuracy_score(y, predictions_clf_01))\n        print(\"\\n\")\n        print(\"confusion matrix is:\", confusion_matrix(y, predictions_clf_01))\n        print(\"\\n\")\n        print(\"fbeta is:\", fbeta_score(y, predictions_clf_01, beta=2))\n        print(\"\\n\")\n        print(classification_report(y, predictions_clf_01))\n\n        # Generate ROC curve values: fpr, tpr, thresholds\n        fpr, tpr, thresholds = roc_curve(y, predictions_clf_prob)\n\n        # Plot ROC curve\n        plt.plot([0, 1], [0, 1], 'k--')\n        plt.plot(fpr, tpr)\n        plt.xlabel('False Positive Rate')\n        plt.ylabel('True Positive Rate')\n        plt.title('ROC Curve')\n        plt.show()\n\n        # Plot non-normalized confusion matrix\n        cls.plot_confusion_matrix(y, predictions_clf_01,\n                                  title='Confusion matrix, without normalization')\n\n        # Plot normalized confusion matrix\n        cls.plot_confusion_matrix(y, predictions_clf_01, normalize=True,\n                                  title='Normalized confusion matrix')\n        plt.show()\n        # Plot Precision Recall Curve\n        precisions, recalls, thresholds = precision_recall_curve(\n            y, predictions_clf_prob)\n        cls.plot_precision_recall_vs_thresholds(\n            precisions, recalls, thresholds)\n        plt.show()\n        # Report\n        kds.metrics.report(y, predictions_clf_prob)\n\n        return predictions_clf_prob, predictions_clf_01\n\n    @classmethod\n    def plot_importance(cls, model, X, num=10):\n        feature_imp = pd.DataFrame(\n            {'Value': model.feature_importance(), 'Feature': X.columns})\n        plt.figure(figsize=(20, 10))\n        sns.set(font_scale=1)\n        sns.barplot(x=\"Value\", y=\"Feature\", data=feature_imp.sort_values(by=\"Value\",\n                                                                         ascending=False)[0:num])\n        plt.title('LightGBM Features (avg over folds)')\n        plt.tight_layout()\n        plt.savefig('lgbm_importance.png')\n        plt.show()\n\ndef iv_woe(data, target, bins=10, show_woe=False):\n    # Empty Dataframe\n    newDF, woeDF = pd.DataFrame(), pd.DataFrame()\n    \n    # Extract Column Names\n    cols = data.columns\n\n    # Run WOE and IV on all the independent variables\n    for ivars in cols[~cols.isin(['case_id', 'WEEK_NUM', target])]:\n        try:\n            if (data[ivars].dtype.kind in 'bifc') and (len(np.unique(data[ivars])) > 20):\n                # Handle missing values before applying pd.qcut\n                binned_x = pd.qcut(data[ivars].dropna().astype('float64'), bins, duplicates='drop')\n                d0 = pd.DataFrame({'x': binned_x, 'y': data[target]})\n            else:\n                d0 = pd.DataFrame({'x': data[ivars], 'y': data[target]})\n                \n            d0 = d0.astype({\"x\": str})\n            d = d0.groupby(\"x\", as_index=False, dropna=False).agg({\"y\": [\"count\", \"sum\"]})\n            d.columns = ['Cutoff', 'N', 'Events']\n            d['% of Events'] = np.maximum(d['Events'], 0.5) / d['Events'].sum()\n            d['Non-Events'] = d['N'] - d['Events']\n            d['% of Non-Events'] = np.maximum(d['Non-Events'], 0.5) / d['Non-Events'].sum()\n            d['WoE'] = np.log(d['% of Non-Events'] / d['% of Events'])\n            d['IV'] = d['WoE'] * (d['% of Non-Events'] - d['% of Events'])\n            d.insert(loc=0, column='Variable', value=ivars)\n            print(\"Information value of \" + ivars + \" is \" + str(round(d['IV'].sum(), 6)))\n            temp = pd.DataFrame({\"Variable\": [ivars], \"IV\": [d['IV'].sum()]}, columns=[\"Variable\", \"IV\"])\n            newDF = pd.concat([newDF, temp], axis=0)\n            woeDF = pd.concat([woeDF, d], axis=0)\n\n            # Show WOE Table\n            if show_woe == True:\n                print(d)\n        except ValueError as ve:\n            print(f\"Skipping {ivars} due to error: {ve}\")\n            continue\n    \n    return newDF, woeDF\n\n\n\ndef woe_categorical(df, cat_feature, good_bad_df):\n    df = pd.concat([df[cat_feature], good_bad_df], axis=1)\n    df = pd.concat([df.groupby(df.columns.values[0], as_index=False)[df.columns.values[1]].count(),\n                    df.groupby(df.columns.values[0], as_index=False)[df.columns.values[1]].mean()], axis=1)\n    df = df.iloc[:, [0, 1, 3]]\n    df.columns = [df.columns.values[0], 'n_obs', 'prop_good']\n    df['prop_n_obs'] = df['n_obs'] / df['n_obs'].sum()\n    df['n_good'] = df['prop_good'] * df['n_obs']\n    df['n_bad'] = (1 - df['prop_good']) * df['n_obs']\n    df['prop_n_good'] = df['n_good'] / df['n_good'].sum()\n    df['prop_n_bad'] = df['n_bad'] / df['n_bad'].sum()\n    df['WoE'] = np.log(df['prop_n_good'] / df['prop_n_bad'])\n    df = df.sort_values(['WoE'])\n    df = df.reset_index(drop=True)\n    df['diff_prop_good'] = df['prop_good'].diff().abs()\n    df['diff_WoE'] = df['WoE'].diff().abs()\n    df['IV'] = (df['prop_n_good'] - df['prop_n_bad']) * df['WoE']\n    df['IV'] = df['IV'].sum()\n    return df\n\n# function to calculate WoE for continous variables\n\n\ndef woe_continuous_multiple(df, cont_feature1, cont_feature2, good_bad_df):\n    df = pd.concat([df[[cont_feature1, cont_feature2]], good_bad_df], axis=1)\n    df['n_obs'] = 1\n    df_grouped = df.groupby([cont_feature1, cont_feature2], as_index=False).agg({\n        'n_obs': 'count',\n        good_bad_df.name: ['mean', 'sum']\n    })\n    df_grouped.columns = [cont_feature1,\n                          cont_feature2, 'n_obs', 'prop_good', 'n_good']\n    df_grouped['prop_n_obs'] = df_grouped['n_obs'] / df_grouped['n_obs'].sum()\n    df_grouped['n_bad'] = df_grouped['n_obs'] - df_grouped['n_good']\n    df_grouped['prop_n_good'] = df_grouped['n_good'] / \\\n        df_grouped['n_good'].sum()\n    df_grouped['prop_n_bad'] = df_grouped['n_bad'] / df_grouped['n_bad'].sum()\n    df_grouped['WoE'] = np.log(\n        df_grouped['prop_n_good'] / df_grouped['prop_n_bad'])\n    df_grouped['diff_prop_good'] = df_grouped['prop_good'].diff().abs()\n    df_grouped['diff_WoE'] = df_grouped['WoE'].diff().abs()\n    df_grouped['IV'] = (df_grouped['prop_n_good'] -\n                        df_grouped['prop_n_bad']) * df_grouped['WoE']\n    df_grouped = df_grouped[[cont_feature1,\n                             cont_feature2, 'n_obs', 'prop_good', 'WoE', 'IV']]\n\n    return df_grouped\n\n\ndef woe_continous(df, cat_feature, good_bad_df):\n    df = pd.concat([df[cat_feature], good_bad_df], axis=1)\n    df = pd.concat([df.groupby(df.columns.values[0], as_index=False)[df.columns.values[1]].count(),\n                    df.groupby(df.columns.values[0], as_index=False)[df.columns.values[1]].mean()], axis=1)\n    df = df.iloc[:, [0, 1, 3]]\n    df.columns = [df.columns.values[0], 'n_obs', 'prop_good']\n    df['prop_n_obs'] = df['n_obs'] / df['n_obs'].sum()\n    df['n_good'] = df['prop_good'] * df['n_obs']\n    df['n_bad'] = (1 - df['prop_good']) * df['n_obs']\n    df['prop_n_good'] = df['n_good'] / df['n_good'].sum()\n    df['prop_n_bad'] = df['n_bad'] / df['n_bad'].sum()\n    df['WoE'] = np.log(df['prop_n_good'] / df['prop_n_bad'])\n    df['diff_prop_good'] = df['prop_good'].diff().abs()\n    df['diff_WoE'] = df['WoE'].diff().abs()\n    df['IV'] = (df['prop_n_good'] - df['prop_n_bad']) * df['WoE']\n    df['IV'] = df['IV'].sum()\n    return df\n\n\ndef plot_by_woe(df_WoE, rotation_of_x_axis_labels=45, save_path=None):\n    x = np.array(df_WoE.iloc[:, 0].apply(str))\n    y = df_WoE['WoE']\n    plt.figure(figsize=(15, 10))\n    plt.plot(x, y, marker='o', color='hotpink', linestyle='dashed', linewidth=3,\n             markersize=18, markeredgecolor='cyan', markerfacecolor='black')\n    plt.xlabel(df_WoE.columns[0])\n    plt.ylabel('Weight of Evidence')\n    plt.title(str('Weight of Evidence by ' + df_WoE.columns[0]))\n    plt.xticks(rotation=rotation_of_x_axis_labels)\n    plt.xticks(fontsize=14)\n    plt.yticks(fontsize=14)\n\n    # Save the plot if save_path is provided\n    if save_path:\n        plt.savefig(save_path, bbox_inches='tight')\n\n    plt.show()\n    \ndef create_ratios(df):\n    df['credamount770A_disbursedcredamount1113A_ratio'] = df['credamount_770A'] / df['disbursedcredamount_1113A']\n    df['credamount770A_maxcredamount590A_ratio'] = df['credamount_770A'] / df['max_credamount_590A']\n    df['lastapprcredamount781A_maxcredamount590A_ratio'] = df['lastapprcredamount_781A'] / df['max_credamount_590A']\n    df['currdebt22A_maxdebt4972A_ratio'] = df['currdebt_22A'] / df['maxdebt4_972A']\n    df['maxdebtoutstand525A_maxtotaloutstanddebtvalue39A_ratio'] = df['max_debtoutstand_525A'] / df['max_totaloutstanddebtvalue_39A']\n    df['annuitynextmonth57A_annuity780A_ratio'] = df['annuitynextmonth_57A'] / df['annuity_780A']\n    df['numinstpaidearly3d3546850L_numinstpaid4499208L_ratio'] = df['numinstpaidearly3d_3546850L'] / df['numinstpaid_4499208L']\n    df['numinstpaidearly3dest4493216L_numinstpaid4499208L_ratio'] = df['numinstpaidearly3dest_4493216L'] / df['numinstpaid_4499208L']\n    df['numinstpaidearly5d1087L_numinstpaid4499208L_ratio'] = df['numinstpaidearly5d_1087L'] / df['numinstpaid_4499208L']\n    df['numinstpaidearly5dest4493211L_numinstpaid4499208L_ratio'] = df['numinstpaidearly5dest_4493211L'] / df['numinstpaid_4499208L']\n    df['numinstpaidearly5dobd4499205L_numinstpaid4499208L_ratio'] = df['numinstpaidearly5dobd_4499205L'] / df['numinstpaid_4499208L']\n    df['numinstpaidearly338L_numinstpaid4499208L_ratio'] = df['numinstpaidearly_338L'] / df['numinstpaid_4499208L']\n    df['numinstpaidearlyest4493214L_numinstpaid4499208L_ratio'] = df['numinstpaidearlyest_4493214L'] / df['numinstpaid_4499208L']\n    df['numinstpaidlastcontr4325080L_numinstpaid4499208L_ratio'] = df['numinstpaidlastcontr_4325080L'] / df['numinstpaid_4499208L']\n    df['numinstpaidlate1d3546852L_numinstpaid4499208L_ratio'] = df['numinstpaidlate1d_3546852L'] / df['numinstpaid_4499208L']\n    \n    return df\n\ndef group_experiment(df, date_column):\n    df = df.copy()\n    # sort values first\n    df = df.sort_values([date_column])\n\n    search_date_df = pd.DataFrame(df[date_column].drop_duplicates())\n    search_date_df = search_date_df.sort_values([date_column]).reset_index()\n    # drop column\n    search_date_df = search_date_df.drop(['index'], axis=1)\n    search_date_df = search_date_df.reset_index()\n    search_date_df.rename(columns={search_date_df.columns[0]: \"order\",\n                                   search_date_df.columns[1]: date_column, }, inplace=True)\n\n    print(search_date_df.columns)\n    print(search_date_df.head(3))\n\n    search_date_df[date_column] = search_date_df[date_column].astype(int)\n    df[date_column] = df[date_column].astype(int)\n\n    # join with data\n    # bring in the order of observations\n    df = pd.merge(df, search_date_df[[date_column, 'order']].drop_duplicates(),\n                  on=[date_column],\n                  how='inner')\n\n    return df\n\ndef visualize_continuous_woe(df, feature):\n    feature_bin = feature + '_bin'\n    df[feature_bin] = pd.qcut(df[feature].dropna().astype('float64'), 10, duplicates='drop')\n    df_woe = woe_continous(\n        df, feature_bin, df_train_encoded['target'])\n    plot_by_woe(df_woe)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:14:19.936078Z","iopub.execute_input":"2024-05-30T23:14:19.936423Z","iopub.status.idle":"2024-05-30T23:14:20.129190Z","shell.execute_reply.started":"2024-05-30T23:14:19.936395Z","shell.execute_reply":"2024-05-30T23:14:20.128237Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Import data with all associated features","metadata":{}},{"cell_type":"code","source":"ROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"\nID_LEVEL = 'case_id'\nDATE_COLUMN = 'WEEK_NUM'\nTARGET = 'target'\nSAMPLE = 0.3\n\ndata_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2)\n    ]\n}\n\ndf_train = feature_eng(**data_store)\ndel data_store\ndf_train = df_train.pipe(Pipeline.filter_cols)\ngc.collect()\ndf_train = df_train.to_pandas()\nprint(\"train data shape:\\t\", df_train.shape)\n# Select columns with object dtype using select_dtypes() method\nobject_columns = df_train.select_dtypes(include=['object']).columns.tolist()\ndf_train = create_ratios(df_train)\ndf_train.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_train = reduce_mem_usage(df_train)\n# get a sample 10% of the total df!\n# Group the DataFrame by the stratification feature\ngrouped = df_train.groupby('WEEK_NUM', group_keys=False)\n# Sample within each group using apply and lambda function\ndf_train = grouped.apply(lambda x: x.sample(frac=SAMPLE, random_state=42))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:14:25.622786Z","iopub.execute_input":"2024-05-30T23:14:25.623387Z","iopub.status.idle":"2024-05-30T23:19:01.120999Z","shell.execute_reply.started":"2024-05-30T23:14:25.623358Z","shell.execute_reply":"2024-05-30T23:19:01.120006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Perform Feature Selection","metadata":{}},{"cell_type":"markdown","source":"Check individual feature predictive power by Information Value per feature. It can be quite informative and give extra ideas for further feature engineering.","metadata":{}},{"cell_type":"markdown","source":"Use LabelEncodingColumns in order to convert all categorical feature to numeric. This way we can inspect and measure feature predictive power.","metadata":{}},{"cell_type":"code","source":"df_train_encoded = df_train.copy()\nencoder = LabelEncodingColumns(object_columns)\nfor col in object_columns:\n    df_train_encoded[col] = df_train_encoded[col].astype(str)\n    \ndf_train_encoded = encoder.fit_transform(df_train_encoded)\n# Set indices\ndf_train_encoded = df_set_indices(df_train_encoded, ID_LEVEL, DATE_COLUMN)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T22:32:06.273809Z","iopub.execute_input":"2024-05-30T22:32:06.274497Z","iopub.status.idle":"2024-05-30T22:32:15.156103Z","shell.execute_reply.started":"2024-05-30T22:32:06.274450Z","shell.execute_reply":"2024-05-30T22:32:15.154754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the features with the most predictive power using the Information Value (IV) metric. The top features can contribute to interpret why and how each feature interacts with credit risk default. ","metadata":{}},{"cell_type":"code","source":"info, woe_map = iv_woe(df_train_encoded, TARGET)\nwoe_map_iv = woe_map.sort_values(by='IV', ascending=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T22:32:15.158005Z","iopub.execute_input":"2024-05-30T22:32:15.158434Z","iopub.status.idle":"2024-05-30T22:33:18.557083Z","shell.execute_reply.started":"2024-05-30T22:32:15.158394Z","shell.execute_reply":"2024-05-30T22:33:18.555959Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"woe_map_iv.head(15)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Choose top x features sorted by IV and plot them vs Weight of Evidence. This way we can get extra insights for the problem in hand and explain it in a more intuitive way.","metadata":{}},{"cell_type":"code","source":"woe_info_sorted = info.sort_values(by='IV', ascending=False)\n# Check the top 15 variables with the highest IV values\nwoe_info_sorted.head(15)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T22:33:48.180204Z","iopub.execute_input":"2024-05-30T22:33:48.180645Z","iopub.status.idle":"2024-05-30T22:33:48.194429Z","shell.execute_reply.started":"2024-05-30T22:33:48.180595Z","shell.execute_reply":"2024-05-30T22:33:48.193238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_to_display = woe_info_sorted.head(15)['Variable'].tolist()\n\nfor feat in features_to_display:\n    visualize_continuous_woe(df_train_encoded, feat)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T22:41:12.188715Z","iopub.execute_input":"2024-05-30T22:41:12.189149Z","iopub.status.idle":"2024-05-30T22:41:17.855103Z","shell.execute_reply.started":"2024-05-30T22:41:12.189117Z","shell.execute_reply":"2024-05-30T22:41:17.853892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we will select just 100 features to fit our LGMB model. We will do it using a voting mechanism which selects features using multiple and different approaches and in the end 100 features will be selected based on this voting mechanism.","metadata":{}},{"cell_type":"code","source":"df_train_encoded = df_train.copy()\nfor col in object_columns:\n    df_train_encoded[col] = df_train_encoded[col].astype(str)\ndf_train_encoded = encoder.fit_transform(df_train_encoded)\n# Set indices\ndf_train_encoded = df_set_indices(df_train_encoded, ID_LEVEL, DATE_COLUMN)\nx_train_encoded, y_train_encoded = create_xy(df_train_encoded, TARGET)\n# Feature Selection\nselected_feature_list = FeatureSelection.variable_selection(x_train_encoded,\n                                                y_train_encoded,\n                                                450,\n                                                100)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T22:49:20.576795Z","iopub.execute_input":"2024-05-30T22:49:20.577771Z","iopub.status.idle":"2024-05-30T22:55:10.661643Z","shell.execute_reply.started":"2024-05-30T22:49:20.577734Z","shell.execute_reply":"2024-05-30T22:55:10.660623Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_feature_list","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"# Scoring Optimization\nSCORING = 'roc_auc'\n\nNUM_OF_EVALS = 5\n\n# Define searched space\nLGBM_HYPER_SPACE = {'objective': 'binary',\n                    'metric': 'auc',\n                    'boosting': 'gbdt',\n                    'silent': True,\n                    \"device\": \"gpu\", \n                    'verbose': -1,\n                    'extra_trees': hp.choice('extra_trees', [True, False]),\n                    'n_estimators': hp.choice('n_estimators', list(range(50, 3000, 200))),\n                    'scale_pos_weight': hp.choice('scale_pos_weight', list(range(25, 33, 1))),\n                    'max_depth': hp.choice('max_depth', list(range(6, 20, 1))),\n                    'num_leaves': hp.choice('num_leaves', list(range(50, 8000, 100))),\n                    'subsample': hp.choice('subsample', [.5, .6, .7, .8, .9, 1]),\n                    'colsample_bytree': hp.uniform('colsample_bytree', 0.3, 1),\n                    'learning_rate': hp.uniform('learning_rate', 0.05, 0.17),\n                    'reg_alpha': hp.uniform('reg_alpha', 0, 1),\n                    'reg_lambda': hp.uniform('reg_lambda', 0, 1),\n                    'min_child_samples': hp.choice('min_child_samples', list(range(20, 400, 40))),\n                    'bagging_fraction': hp.uniform('bagging_fraction', 0.5, 1),\n                    'feature_fraction': hp.uniform('feature_fraction', 0.5, 1),\n                    'min_split_gain': hp.uniform('min_split_gain', 0, 0.5),\n                    }\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:19:01.122876Z","iopub.execute_input":"2024-05-30T23:19:01.123221Z","iopub.status.idle":"2024-05-30T23:19:01.134649Z","shell.execute_reply.started":"2024-05-30T23:19:01.123191Z","shell.execute_reply":"2024-05-30T23:19:01.133695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = group_experiment(df_train, 'WEEK_NUM')\ntime_train = df_train['order'].copy()\ndf_train[object_columns] = df_train[object_columns].astype(\"category\")    \n# Set indices\ndf_train_ind = df_set_indices(df_train, ID_LEVEL, DATE_COLUMN)\nx_train, y_train = create_xy(df_train_ind, TARGET)\n# Keep Selected Features\nx_train = x_train[selected_feature_list]","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:19:01.154969Z","iopub.execute_input":"2024-05-30T23:19:01.155252Z","iopub.status.idle":"2024-05-30T23:19:01.788735Z","shell.execute_reply.started":"2024-05-30T23:19:01.155221Z","shell.execute_reply":"2024-05-30T23:19:01.787694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will perform Time Based Cross Validation with respect to the week number in the training dataset. Using time based cross val we will search for the hyperparameter space to find the optimal values of the LGBM model and then fit the model.","metadata":{}},{"cell_type":"code","source":"# Choose Best Lightgbm model\nlgbm_chosen, best_params = LightgbmPipeline.pipeline(x_train,\n                                                     y_train,\n                                                     LGBM_HYPER_SPACE,\n                                                     SCORING,\n                                                     time_train,\n                                                     NUM_OF_EVALS)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:19:01.789991Z","iopub.execute_input":"2024-05-30T23:19:01.790323Z","iopub.status.idle":"2024-05-30T23:24:14.291247Z","shell.execute_reply.started":"2024-05-30T23:19:01.790298Z","shell.execute_reply":"2024-05-30T23:24:14.289544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"PlotMetrics.plot_predictions(x_train, y_train, lgbm_chosen)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:25:10.143119Z","iopub.execute_input":"2024-05-30T23:25:10.143407Z","iopub.status.idle":"2024-05-30T23:25:14.750684Z","shell.execute_reply.started":"2024-05-30T23:25:10.143383Z","shell.execute_reply":"2024-05-30T23:25:14.749748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PlotMetrics.plot_importance(lgbm_chosen, x_train, num=15)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:25:14.751929Z","iopub.execute_input":"2024-05-30T23:25:14.752278Z","iopub.status.idle":"2024-05-30T23:25:14.873249Z","shell.execute_reply.started":"2024-05-30T23:25:14.752247Z","shell.execute_reply":"2024-05-30T23:25:14.872048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del y_train, df_train, df_train_ind","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:24:14.295837Z","iopub.status.idle":"2024-05-30T23:24:14.296167Z","shell.execute_reply.started":"2024-05-30T23:24:14.296000Z","shell.execute_reply":"2024-05-30T23:24:14.296013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store_test = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}\n\ndf_test = feature_eng(**data_store_test)\nprint(\"test data shape:\\t\", df_test.shape)\n\ndf_test = df_test.to_pandas()\ndf_test = create_ratios(df_test)\ndf_test = reduce_mem_usage(df_test)\n# Set indices\ndf_test = df_test.drop(columns=[DATE_COLUMN])\ndf_test = df_test.set_index(ID_LEVEL)\n# Keep Selected Features\nx_test = df_test[selected_feature_list]\nlgb_pred = lgbm_chosen.predict(x_test)\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\ndf_subm[\"score\"] = lgb_pred\ndf_subm","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:28:03.885511Z","iopub.execute_input":"2024-05-30T23:28:03.885994Z","iopub.status.idle":"2024-05-30T23:28:04.913962Z","shell.execute_reply.started":"2024-05-30T23:28:03.885957Z","shell.execute_reply":"2024-05-30T23:28:04.912950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have trained a LightGBM model named 'lgb_model'\n\n# Specify the file path where you want to save the model\nmodel_file_path = \"lgb_model_v7.pkl\"  # You can change the file extension as needed\n\n# Save the model using joblib\njoblib.dump(lgbm_chosen, model_file_path)\n\nprint(\"Model saved successfully to:\", model_file_path)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}