{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# My Notebook\n\n## Tasks: \n1. Choose features\n2. implement feature engineering\n3. optimize memory\n4. hyperpamater tuning\n5. multi lgm models, choosing best approach","metadata":{}},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt \nfrom sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\n\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.impute import KNNImputer\n\n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:30:28.235886Z","iopub.execute_input":"2024-05-27T06:30:28.236272Z","iopub.status.idle":"2024-05-27T06:30:31.879151Z","shell.execute_reply.started":"2024-05-27T06:30:28.236243Z","shell.execute_reply":"2024-05-27T06:30:31.877923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Preprocessing:\n\n    # Setting correct types for less storage space and faster execution \n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n        #Basically we want to check delta of decision date and prior date (date of request e.g)\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\")) \n                df = df.with_columns(pl.col(col).dt.total_days()) # Extracting total days from {numerical} + D\n        # Month is a numerical variable, not informative, We used date_decision        \n        df = df.drop(\"date_decision\") \n        return df\n\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]: #Crucial columns, especially target \n                isnull = df[col].is_null().mean()\n                if isnull > 0.7:  # For effeciency and feature engineering\n                    df = df.drop(col)\n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200): # For effeciency and feature engineering\n                    df = df.drop(col)\n        \n        return df\n\n\nclass Aggregator:\n    #For each type we set featues last, max, mean:\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        \n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        # New columns for max, last instance in a group and mean of columns\n        return expr_max +expr_last+expr_mean\n    \n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        return  expr_max +expr_last+expr_mean\n    \n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        #expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n        return  expr_max +expr_last#+expr_count\n    \n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return  expr_max +expr_last\n    \n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] \n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return  expr_max +expr_last\n    \n    def get_exprs(df): #Get new features, a function for easier call for all permutations\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n        return exprs\n\ndef read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Preprocessing.set_table_dtypes)\n    if depth in [1,2]: #Adding mean,max,last only for more than 1 period\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df)) \n    return df\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    \n    for path in glob(str(regex_path)): #Going thru all files in an effecient way \n        df = pl.read_parquet(path)\n        df = df.pipe(Preprocessing.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df\n\n#different lags, for all depths , and only for df_base which is interesting\ndef feature_eng(df_base, depth_0, depth_1, depth_2): \n    df_base = (\n        df_base\n        .with_columns(\n            #month_decision = np.sin(np.pi * pl.col(\"date_decision\").dt.month() / 30).cast(pl.Int16),\n            #weekday_decision = np.sin(np.pi * pl.col(\"date_decision\").dt.weekday() / 7).cast(pl.Int16),\n            dayOfYear_decision = pl.col(\"date_decision\").dt.ordinal_day(),\n            #month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n            day_decision = pl.col(\"date_decision\").dt.day(),\n            \n        )\n    )\n    df_base = df_base.with_columns(\n        #dayOfYear_sin = np.sin(np.pi * pl.col(\"dayOfYear_decision\") / 183),\n        #weekday_sin = np.sin(np.pi * pl.col(\"weekday_decision\") / 7),\n        #dayOfYear_cos = np.cos(np.pi * pl.col(\"dayOfYear_decision\") / 183),\n        #weekday_cos = np.cos(np.pi * pl.col(\"weekday_decision\") / 7)\n    )\n    #df_base.drop(columns = ['dayOfYear_decision','weekday_decision'])\n     #Join all depths on case_id\n    for i, df  in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n                                  \n    df_base = df_base.pipe(Preprocessing.handle_dates)\n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):  #Deal with object columns\n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols\n\n##NEED TO CHECK WHAT IT MEANS\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:30:44.681474Z","iopub.execute_input":"2024-05-27T06:30:44.681907Z","iopub.status.idle":"2024-05-27T06:30:44.728958Z","shell.execute_reply.started":"2024-05-27T06:30:44.681874Z","shell.execute_reply":"2024-05-27T06:30:44.727491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:30:49.891341Z","iopub.execute_input":"2024-05-27T06:30:49.892633Z","iopub.status.idle":"2024-05-27T06:30:49.898321Z","shell.execute_reply.started":"2024-05-27T06:30:49.892584Z","shell.execute_reply":"2024-05-27T06:30:49.897100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2)\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:30:51.741853Z","iopub.execute_input":"2024-05-27T06:30:51.742711Z","iopub.status.idle":"2024-05-27T06:33:48.231226Z","shell.execute_reply.started":"2024-05-27T06:30:51.742672Z","shell.execute_reply":"2024-05-27T06:33:48.230053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store1 = data_store.copy()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:38:19.265963Z","iopub.execute_input":"2024-05-27T06:38:19.266520Z","iopub.status.idle":"2024-05-27T06:38:19.272677Z","shell.execute_reply.started":"2024-05-27T06:38:19.266482Z","shell.execute_reply":"2024-05-27T06:38:19.271391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store1)\nprint(\"train data shape:\\t\", df_train.shape)\n# Managing storage \ndel data_store1\ngc.collect()\ndf_train = df_train.pipe(Preprocessing.filter_cols)\ndf_train, cat_cols = to_pandas(df_train)\ndf_train = reduce_mem_usage(df_train)\nprint(\"train data shape:\\t\", df_train.shape)\nnums=df_train.select_dtypes(exclude='category').columns\nfrom itertools import combinations, permutations\n#df_train=df_train[nums]\nnans_df = df_train[nums].isna()\nnans_groups={}\nfor col in nums:\n    cur_group = nans_df[col].sum()\n    try:\n        nans_groups[cur_group].append(col)\n    except:\n        nans_groups[cur_group]=[col]\ndel nans_df; x=gc.collect()\n\ndef reduce_group(grps):\n    use = []\n    for g in grps:\n        mx = 0; vx = g[0]\n        for gg in g:\n            n = df_train[gg].nunique()\n            if n>mx:\n                mx = n\n                vx = gg\n            #print(str(gg)+'-'+str(n),', ',end='')\n        use.append(vx)\n        #print()\n    print('Use these',use)\n    return use\n\ndef group_columns_by_correlation(matrix, threshold=0.8):\n    # 计算列之间的相关性\n    correlation_matrix = matrix.corr()\n\n    # 分组列\n    groups = []\n    remaining_cols = list(matrix.columns)\n    while remaining_cols:\n        col = remaining_cols.pop(0)\n        group = [col]\n        correlated_cols = [col]\n        for c in remaining_cols:\n            if correlation_matrix.loc[col, c] >= threshold:\n                group.append(c)\n                correlated_cols.append(c)\n        groups.append(group)\n        remaining_cols = [c for c in remaining_cols if c not in correlated_cols]\n    \n    return groups\n\nuses=[]\nfor k,v in nans_groups.items():\n    if len(v)>1:\n            Vs = nans_groups[k]\n            #cross_features=list(combinations(Vs, 2))\n            #make_corr(Vs)\n            grps= group_columns_by_correlation(df_train[Vs], threshold=0.8)\n            use=reduce_group(grps)\n            uses=uses+use\n            #make_corr(use)\n    else:\n        uses=uses+v\n    print('####### NAN count =',k)\nprint(uses)\nprint(len(uses))\nuses=uses+list(df_train.select_dtypes(include='category').columns)\nprint(len(uses))\ndf_train=df_train[uses]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:38:20.848394Z","iopub.execute_input":"2024-05-27T06:38:20.848810Z","iopub.status.idle":"2024-05-27T06:40:33.366219Z","shell.execute_reply.started":"2024-05-27T06:38:20.848778Z","shell.execute_reply":"2024-05-27T06:40:33.364899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train1 = df_train.copy()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:33.368842Z","iopub.execute_input":"2024-05-27T06:40:33.369901Z","iopub.status.idle":"2024-05-27T06:40:34.312315Z","shell.execute_reply.started":"2024-05-27T06:40:33.369849Z","shell.execute_reply":"2024-05-27T06:40:34.311153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train1['dayOfYear_sin'] = np.sin(np.pi* df_train1['dayOfYear_decision'] / 183)\ndf_train1['weekday_sin'] = np.sin(np.pi* df_train1['weekday_decision'] / 7)\ndf_train1['dayOfYear_cos'] = np.cos(np.pi* df_train1['dayOfYear_decision'] / 183)\ndf_train1['weekday_cos'] = np.cos(np.pi* df_train1['weekday_decision'] / 7)\ndf_train1.drop(columns = ['dayOfYear_decision','weekday_decision'],inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:34.313971Z","iopub.execute_input":"2024-05-27T06:40:34.314451Z","iopub.status.idle":"2024-05-27T06:40:35.592998Z","shell.execute_reply.started":"2024-05-27T06:40:34.314415Z","shell.execute_reply":"2024-05-27T06:40:35.591803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train1['dayOfYear_sin']","metadata":{"execution":{"iopub.status.busy":"2024-05-26T11:52:30.882798Z","iopub.execute_input":"2024-05-26T11:52:30.883330Z","iopub.status.idle":"2024-05-26T11:52:30.895419Z","shell.execute_reply.started":"2024-05-26T11:52:30.883284Z","shell.execute_reply":"2024-05-26T11:52:30.894073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#file = pl.read_parquet(TRAIN_DIR / \"train_static_cb_0.parquet\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\" \ndef date_m(df):\n# Filter columns that end with 'D'\n    sin_d,cos_d,sin_w,cos_w = [],[],[],[]\n    date_cols = [col for col in df.columns if col.endswith('D')]\n    \n    for col in date_cols:\n        #date_cols = date_cols.with_columns(pl.col(col).cast(pl.Date))\n        # Compute day_of_year, hour, and weekday for all columns at once\n        date_col = pl.col(col).cast(pl.Date)\n        day_of_yeare = date_col.dt.ordinal_day() #.alias(f\"dayofy_{col}\") \n        #hour = pl.max(df[date_cols]).dt.hour()\n        weekdaye = date_col.dt.weekday().alias(f\"weekday_{col}\") \n        sin_day = (np.pi * day_of_yeare / 183).sin().max().alias(f\"sin(dayofyear)_{col}\")\n        cos_day = (np.pi * day_of_yeare / 183).cos().max().alias(f\"cos(dayofyear)_{col}\")\n        sin_week = (np.pi * weekdaye / 7).sin().max().alias(f\"sin(week)_{col}\")\n        cos_week = (np.pi * weekdaye / 7).cos().max().alias(f\"cos(week)_{col}\")\n        sin_d.append(sin_day)\n        cos_d.append(cos_day)\n        sin_w.append(sin_week)\n        cos_w.append(cos_week)\n        #print(cos_w)\n        # Compute sine and cosine values for all columns at once\n        exp = sin_d + cos_d + sin_w + cos_w\n        #(np.pi * day_of_year / 183).sin().alias(f\"sin_dayofyear\") + \\\n        #(np.pi * day_of_year / 183).cos().alias(f\"cos_dayofyear\") + \\\n        #(np.pi * weekday / 7).sin().alias(f\"sin_weekday\") + \\\n        #(np.pi * weekday / 7).cos().alias(f\"cos_weekday\")\n\n    return exp\n    \"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\"\nmerged_df = file1.join(file2, on=\"case_id\", how=\"inner\")\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ndef get_specific_column(data_store, key, column_index):\n    specific_columns = []\n    for df in data_store[key]:\n        if isinstance(df, list):\n            for sub_df in df:\n                if column_index < len(sub_df.columns):\n                    specific_columns.append(sub_df[:, column_index])\n        else:\n            if column_index < len(df.columns):\n                specific_columns.append(df[:, column_index])\n    return specific_columns\n\n# Get the 5th column from each DataFrame in depth_0\ndepth_0_column_5 = get_specific_column(data_store, 'depth_0',110)\ndepth_0_column_5\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# Function to check for columns with 'sin' in their names in depth_0\ndef find_columns_with_sin(data_store, key):\n    for df in data_store[key]:\n        if isinstance(df, list):\n            for sub_df in df:\n                sin_columns = [col for col in sub_df.columns if 'maxNEW' in col]\n                if sin_columns:\n                    return True, sin_columns\n        else:\n            sin_columns = [col for col in df.columns if 'last' in col]\n            if sin_columns:\n                return True, sin_columns\n    return False, []\n\n# Check for columns with 'sin' in their names in depth_0\nhas_sin_columns, sin_columns = find_columns_with_sin(data_store, 'depth_1')\n\n# Print the result\nif has_sin_columns:\n    print(\"Columns with 'sin' found:\", sin_columns)\nelse:\n    print(\"No columns with 'sin' found in depth_0.\")\n\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ndef get_columns_from_depth(depth_data):\n    columns = {}\n    for idx, df in enumerate(depth_data):\n        # Ensure that each item in depth_data is a DataFrame\n        if isinstance(df, list):\n            for sub_idx, sub_df in enumerate(df):\n                columns[f\"depth_1{idx}_{sub_idx}\"] = sub_df.columns\n        else:\n            columns[f\"depth_1{idx}\"] = df.columns\n    return columns\n\n# Use the function to get columns from depth_2\ndepth_1_columns = get_columns_from_depth(data_store['depth_1'])\n\n# Print the columns\nfor key, cols in depth_1_columns.items():\n    print(f\"{key}: {cols}\")\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n# Function to check for object columns in a dataframe and print them\ndef print_object_columns(df_name, df):\n    object_columns = [col for col in df.columns if df[col].dtype == 'object']\n    print(f\"Debug: DataFrame '{df_name}' has {len(object_columns)} object columns.\")\n    if object_columns:\n        print(f\"DataFrame '{df_name}' has object columns:\")\n        for col in object_columns:\n            print(f\"- {col}\")\n\n# Iterate through the data_store dictionary\nfor key, value in data_store.items():\n    # If the value is a single dataframe\n    if isinstance(value, pd.DataFrame):\n        print_object_columns(key, value)\n    # If the value is a list of dataframes\n    elif isinstance(value, list):\n        for i, df in enumerate(value):\n            print_object_columns(f\"{key}[{i}]\", df)\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\ndevice='gpu'\n#n_samples=200000\nn_est=6000\nDRY_RUN = True if sample.shape[0] == 10 else False   \nif DRY_RUN:\n    device='cpu'\n    df_train1 = df_train1.iloc[:50000]\n    #n_samples=10000\n    n_est=600\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:35.598902Z","iopub.execute_input":"2024-05-27T06:40:35.599314Z","iopub.status.idle":"2024-05-27T06:40:35.619639Z","shell.execute_reply.started":"2024-05-27T06:40:35.599279Z","shell.execute_reply":"2024-05-27T06:40:35.618659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:35.621028Z","iopub.execute_input":"2024-05-27T06:40:35.621601Z","iopub.status.idle":"2024-05-27T06:40:36.864615Z","shell.execute_reply.started":"2024-05-27T06:40:35.621545Z","shell.execute_reply":"2024-05-27T06:40:36.863584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\ndf_test = feature_eng(**data_store)\nprint(\"test data shape:\\t\", df_test.shape)\ndel data_store\ngc.collect()\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)\n\ndf_test, cat_cols = to_pandas(df_test, cat_cols)\ndf_test = reduce_mem_usage(df_test)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:36.866210Z","iopub.execute_input":"2024-05-27T06:40:36.866618Z","iopub.status.idle":"2024-05-27T06:40:37.523479Z","shell.execute_reply.started":"2024-05-27T06:40:36.866587Z","shell.execute_reply":"2024-05-27T06:40:37.522315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test1 = df_test.copy()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:37.524669Z","iopub.execute_input":"2024-05-27T06:40:37.524994Z","iopub.status.idle":"2024-05-27T06:40:37.540636Z","shell.execute_reply.started":"2024-05-27T06:40:37.524965Z","shell.execute_reply":"2024-05-27T06:40:37.539352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test1['dayOfYear_sin'] = np.sin(np.pi* df_test1['dayOfYear_decision'] / 183)\ndf_test1['weekday_sin'] = np.sin(np.pi* df_test1['weekday_decision'] / 7)\ndf_test1['dayOfYear_cos'] = np.cos(np.pi* df_test1['dayOfYear_decision'] / 183)\ndf_test1['weekday_cos'] = np.cos(np.pi* df_test1['weekday_decision'] / 7)\ndf_test1.drop(columns = ['dayOfYear_decision','weekday_decision'],inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:37.542711Z","iopub.execute_input":"2024-05-27T06:40:37.543161Z","iopub.status.idle":"2024-05-27T06:40:37.558782Z","shell.execute_reply.started":"2024-05-27T06:40:37.543120Z","shell.execute_reply":"2024-05-27T06:40:37.557715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test1.drop(columns = [\"case_id\", \"WEEK_NUM\"],inplace = True)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:05:09.586100Z","iopub.execute_input":"2024-05-27T07:05:09.587523Z","iopub.status.idle":"2024-05-27T07:05:09.601665Z","shell.execute_reply.started":"2024-05-27T07:05:09.587464Z","shell.execute_reply":"2024-05-27T07:05:09.600461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df_train1[\"target\"]\nweeks = df_train1[\"WEEK_NUM\"]\ndf_train= df_train1.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:37.560718Z","iopub.execute_input":"2024-05-27T06:40:37.561042Z","iopub.status.idle":"2024-05-27T06:40:37.688194Z","shell.execute_reply.started":"2024-05-27T06:40:37.561014Z","shell.execute_reply":"2024-05-27T06:40:37.687038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(weeks)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:40:37.691828Z","iopub.execute_input":"2024-05-27T06:40:37.692293Z","iopub.status.idle":"2024-05-27T06:40:38.204136Z","shell.execute_reply.started":"2024-05-27T06:40:37.692248Z","shell.execute_reply":"2024-05-27T06:40:38.202989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,  \n    \"learning_rate\": 0.05,\n    \"n_estimators\": 2000,  \n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.1,\n    \"reg_lambda\": 10,\n    \"extra_trees\":True,\n    'num_leaves':64,\n    \"device\": device, \n    \"verbose\": -1,\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom lightgbm import Dataset\nfor idx_train, idx_valid in cv.split(df_train, y, groups=weeks):#\n    X_train, y_train = df_train.iloc[idx_train], y.iloc[idx_train]# \n    X_valid, y_valid = df_train.iloc[idx_valid], y.iloc[idx_valid]\n    X_train1 = X_train.copy()\n    X_valid1 = X_valid.copy()\n    X_train1[cat_cols] = X_train[cat_cols].astype(\"category\")\n    X_valid1[cat_cols] = X_valid[cat_cols].astype(\"category\")\n    dval = Dataset(X_valid1, label=y_valid, feature_name=cat_cols, categorical_feature=cat_cols)\n    dtrain = Dataset(X_train1, label=y_train, feature_name=cat_cols, categorical_feature=cat_cols)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params= {'reg_alpha': 0.10871784284220098, 'subsample': 0.6364076476535482, 'scale_pos_weight': 0.9269468287381711, 'reg_lambda': 10, 'max_depth': 4, 'eta': 0.02010690445058008, 'colsample_bytree': 0.5825234754691985, 'colsample_bynode': 0.53724917896386}\n#Best RMSE: 0.17707973882973993","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna as optuna\nfrom lightgbm import LGBMRegressor\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom lightgbm import Dataset\nimport warnings\nfrom sklearn.metrics import roc_auc_score\nfrom optuna.samplers import TPESampler, NSGAIISampler, CmaEsSampler, GridSampler\n\nrandom_state =  42\nimport logging\n\n# Set LightGBM logging level to suppress warnings\nlogging.getLogger('lightgbm').setLevel(logging.ERROR)\nwarnings.filterwarnings(\"ignore\")\ndef objective(trial):\n    params = {\n            \"boosting_type\": \"gbdt\",\n            \"objective\": \"binary\",\n            \"metric\": \"auc\",\n            'num_leaves': trial.suggest_int('num_leaves', 56,66),\n            \"extra_trees\":True,\n            \"reg_alpha\": trial.suggest_float('reg_alpha', 0.09, 0.11, log = True),\n            \"n_estimators\": 2000,\n            \"metric\": \"auc\",\n            'subsample': trial.suggest_float('subsample', 0.61, 0.65, log = True),\n            'scale_pos_weight': trial.suggest_float('scale_pos_weight', 0.95, 1.1, log = True),\n            'reg_lambda': 10,\n            #'min_child_weight': 1,\n            'max_depth': trial.suggest_int('max_depth',1,4),\n            #'gamma': trial.suggest_float('gamma', 0.4, 0.88, log = True),\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.05, log = True),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.55, 0.63, log = True),\n            'colsample_bynode': trial.suggest_float('colsample_bynode', 0.55, 0.63, log = True),\n            \"device\": device}\n    for idx_train, idx_valid in cv.split(df_train, y, groups=weeks):\n        X_train, y_train = df_train.iloc[idx_train], y.iloc[idx_train]\n        X_valid, y_valid = df_train.iloc[idx_valid], y.iloc[idx_valid]\n        \n        # Define LightGBM datasets with free_raw_data=False\n        train_data = lgb.Dataset(X_train, label=y_train, categorical_feature=cat_cols, free_raw_data=False)\n        valid_data = lgb.Dataset(X_valid, label=y_valid, categorical_feature=cat_cols, free_raw_data=False)\n        \n        model = lgb.train(params, train_data, valid_sets=[valid_data])\n        predictions = model.predict(X_valid)\n        roc = roc_auc_score(y_valid, predictions)\n        return roc","metadata":{"execution":{"iopub.status.busy":"2024-05-26T12:13:55.198174Z","iopub.execute_input":"2024-05-26T12:13:55.198665Z","iopub.status.idle":"2024-05-26T12:13:56.775735Z","shell.execute_reply.started":"2024-05-26T12:13:55.198622Z","shell.execute_reply":"2024-05-26T12:13:56.774683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nsampler = TPESampler(seed=44)\nstudy = optuna.create_study(direction = 'maximize')\n\n# Optimize the objective function\nstudy.optimize(objective, n_trials=15)\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nfamPArams = study.best_params\nprint(\"Best Hyperparameters:\", famPArams)\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\ndef objective2(trial):\n    params = {\n        \"boosting_type\": \"gbdt\",\n        \"objective\": \"binary\",\n        \"metric\": \"auc\",\n        'num_leaves': trial.suggest_int('num_leaves', 56, 66),\n        \"extra_trees\": True,\n        \"reg_alpha\": trial.suggest_float('reg_alpha', 0.1, 0.2, log=True),\n        \"n_estimators\": 2000,\n        \"subsample\": trial.suggest_float('subsample', 0.6, 0.65, log=True),\n        'scale_pos_weight': trial.suggest_float('scale_pos_weight', 0.95, 1.1, log=True),\n        'reg_lambda': 10,\n        'max_depth': trial.suggest_int('max_depth', 1, 5),\n        'learning_rate': 0.05,\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.7, 0.8, log=True),\n        'colsample_bynode': trial.suggest_float('colsample_bynode', 0.7, 0.8, log=True),\n        \"device\": device\n    }\n    \n    for idx_train, idx_valid in cv.split(df_train, y, groups=weeks):\n        X_train, y_train = df_train.iloc[idx_train], y.iloc[idx_train]\n        X_valid, y_valid = df_train.iloc[idx_valid], y.iloc[idx_valid]\n        \n        # Define LightGBM datasets with free_raw_data=False\n        train_data = lgb.Dataset(X_train, label=y_train, categorical_feature=cat_cols, free_raw_data=False)\n        valid_data = lgb.Dataset(X_valid, label=y_valid, categorical_feature=cat_cols, free_raw_data=False)\n        \n        model = lgb.train(params, train_data, valid_sets=[valid_data])\n        predictions = model.predict(X_valid)\n        roc = roc_auc_score(y_valid, predictions)\n        \n        return roc\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nfixed_params = {\n         \"n_estimators\": 2000,\"extra_trees\":True,'reg_lambda': 10,\n    'learning_rate': 0.05,\"device\": device,'num_leaves': 65,}\nstudy = optuna.create_study(direction = 'maximize')\npartial_sampler = optuna.samplers.PartialFixedSampler(fixed_params, study.sampler)\nstudy.sampler = partial_sampler\nstudy.optimize(objective2, n_trials=15)\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nfamPArams = study.best_params\nprint(\"Best Hyperparameters:\", famPArams)\nall_mParams = {**params, **famPArams}\nprint(\"All Hyperparameters:\", all_mParams)\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\ndef objective2_01(trial):\n    params = {\n        \"boosting_type\": \"gbdt\",\n        \"objective\": \"binary\",\n        \"metric\": \"auc\",\n        'num_leaves': 64,\n        \"extra_trees\": True,\n        \"reg_alpha\": trial.suggest_float('reg_alpha', 0.17, 0.2, log=True),\n        \"n_estimators\": 2000,\n        \"subsample\": trial.suggest_float('reg_alpha', 0.67, 0.75, log=True),\n        'scale_pos_weight': 1,\n        'verbose' : -1,\n        'reg_lambda': 10,\n        'max_depth': 10,\n        'learning_rate': trial.suggest_float('learning_rate', 0.02,0.08),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.68, 0.77, log=True),\n        'colsample_bynode': trial.suggest_float('colsample_bynode', 0.65, 0.76, log=True),\n        \"device\": device\n    }\n    for idx_train, idx_valid in cv.split(df_train, y, groups=weeks):\n        X_train, y_train = df_train.iloc[idx_train], y.iloc[idx_train]\n        X_valid, y_valid = df_train.iloc[idx_valid], y.iloc[idx_valid]\n        \n        # Define LightGBM datasets with free_raw_data=False\n        train_data = lgb.Dataset(X_train, label=y_train, categorical_feature=cat_cols, free_raw_data=False)\n        valid_data = lgb.Dataset(X_valid, label=y_valid, categorical_feature=cat_cols, free_raw_data=False)\n        \n        model = lgb.train(params, train_data, valid_sets=[valid_data])\n        predictions = model.predict(X_valid)\n        roc = roc_auc_score(y_valid, predictions)\n        \n        return roc","metadata":{"execution":{"iopub.status.busy":"2024-05-26T15:11:41.459144Z","iopub.execute_input":"2024-05-26T15:11:41.459660Z","iopub.status.idle":"2024-05-26T15:11:41.472695Z","shell.execute_reply.started":"2024-05-26T15:11:41.459620Z","shell.execute_reply":"2024-05-26T15:11:41.471243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"{'boosting_type': 'gbdt', 'objective': 'binary', 'metric': 'auc', 'max_depth': 10,\n 'learning_rate': 0.028541149782907546, 'n_estimators': 2000, 'colsample_bytree': 0.7415159721237228, \n 'colsample_bynode': 0.7294712313122086, 'verbose': -1, 'random_state': 42,\n         'reg_alpha': 0.1752716267325066, 'reg_lambda': 10, 'extra_trees': True, \n         'num_leaves': 64, 'device': 'cpu'}","metadata":{"execution":{"iopub.status.busy":"2024-05-26T13:05:47.827596Z","iopub.execute_input":"2024-05-26T13:05:47.828404Z","iopub.status.idle":"2024-05-26T13:05:47.837960Z","shell.execute_reply.started":"2024-05-26T13:05:47.828359Z","shell.execute_reply":"2024-05-26T13:05:47.836993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nfixed_params2 = {\n         \"n_estimators\": 2000,\"extra_trees\":True,'reg_lambda': 10,'max_depth': 10,'verbose' : -1,\n    \"device\": device,'num_leaves': 64}\nstudy = optuna.create_study(direction = 'maximize')\npartial_sampler = optuna.samplers.PartialFixedSampler(fixed_params2, study.sampler)\nstudy.sampler = partial_sampler\nstudy.optimize(objective2_01, n_trials=5)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-05-26T15:11:45.224273Z","iopub.execute_input":"2024-05-26T15:11:45.224700Z","iopub.status.idle":"2024-05-26T15:20:00.762811Z","shell.execute_reply.started":"2024-05-26T15:11:45.224665Z","shell.execute_reply":"2024-05-26T15:20:00.761439Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nfamPArams = study.best_params\nprint(\"Best Hyperparameters:\", famPArams)\nall_mParams2 = {**fixed_params2, **famPArams}\nprint(\"All Hyperparameters:\", all_mParams2)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-05-26T15:20:10.809157Z","iopub.execute_input":"2024-05-26T15:20:10.810278Z","iopub.status.idle":"2024-05-26T15:20:10.818235Z","shell.execute_reply.started":"2024-05-26T15:20:10.810225Z","shell.execute_reply":"2024-05-26T15:20:10.816702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nprint('Best hyperparameters:', study.best_params)\nprint('Best RMSE:', study.best_value)\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params =  {'num_leaves': 65, 'reg_alpha': 0.11280656899993056,\n                'subsample': 0.6339677409660418, 'scale_pos_weight': 0.9888400698564654,\n                'max_depth': 2, 'learning_rate': 0.05, 'colsample_bytree': 0.5953245510591638, \n                'colsample_bynode': 0.5848329501339482, \"boosting_type\": \"gbdt\",\"objective\": \"binary\",\n               \"metric\": \"auc\",\"n_estimators\": 2000, \"random_state\": 42,\"metric\": \"auc\" }\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,  \n    \"learning_rate\": 0.05,\n    \"n_estimators\": 2000,  \n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.1,\n    \"reg_lambda\": 10,\n    \"extra_trees\":True,\n    'num_leaves':64,\n    \"device\": device, \n    \"verbose\": -1,\n}\nnew_params =  {'boosting_type': 'gbdt', 'objective': 'binary', \n                      'metric': 'auc', 'max_depth': 2, 'learning_rate': 0.05,\n                      'n_estimators': 2000, 'colsample_bytree': 0.7407515202738159, \n                      'colsample_bynode': 0.7176566545736501, 'random_state': 42, \n                      'reg_alpha': 0.17680059967462355, 'reg_lambda': 10, 'extra_trees': True, \n               'num_leaves': 65, 'device': device, 'subsample': 0.6337243904118188,'verbose':-1,\n                      'scale_pos_weight': 1.0098990527371845}\nnewest_params = {'boosting_type': 'gbdt', 'objective': 'binary', \n                 'metric': 'auc', 'max_depth': 10, 'learning_rate': 0.05,\n                 'n_estimators': 2000, 'colsample_bytree': 0.7511695296814248, \n                 'colsample_bynode': 0.738402615591555, 'verbose': -1, \n                 'random_state': 42, 'reg_alpha': 0.19952063420687474, \n                 'reg_lambda': 10, 'extra_trees': True, \n                 'num_leaves': 64, 'device': device, 'subsample': 0.6994628841133421}\nparams_3_0 = {'boosting_type': 'gbdt', 'objective': 'binary', 'metric': 'auc', 'max_depth': 10,\n              'learning_rate': 0.021285409528001467, 'n_estimators': 2000,\n              'colsample_bytree': 0.7640228673220234, \n              'colsample_bynode': 0.7427956695482733, 'verbose': -1, 'random_state': 42, \n              'reg_alpha': 0.18722930632228799, 'reg_lambda': 10, 'extra_trees': True, 'num_leaves': 64, \n              'device': device, 'subsample': 0.7068992746108957}\nparams_4_0 = {'boosting_type': 'gbdt', 'objective': 'binary', 'metric': 'auc', 'max_depth': 10,\n              'learning_rate': 0.03440286931776178, 'n_estimators': 2000, 'colsample_bytree': 0.7553194257799734,\n              'colsample_bynode': 0.7345699722411386, 'verbose': -1, 'random_state': 42, 'reg_alpha': 0.18758196874589023, 'reg_lambda': 10,\n              'extra_trees': True, 'num_leaves': 64, 'device': device, 'subsample': 0.7194935839076688}\nparams_5_0 = {'boosting_type': 'gbdt', 'objective': 'binary', 'metric': 'auc', 'max_depth': 10, \n              'learning_rate': 0.021171987604055423, 'n_estimators': 2000,\n              'colsample_bytree': 0.7310417717772928, 'colsample_bynode': 0.7012555909502968, \n              'verbose': -1, 'random_state': 42, 'reg_alpha': 0.18954041071816444,\n              'reg_lambda': 10, 'extra_trees': True, 'num_leaves': 64, 'device': device,\n              'subsample': 0.6827473524459137}\nparams_6_0 = {'boosting_type': 'gbdt', 'objective': 'binary', 'metric': 'auc', 'max_depth': 10,\n              'learning_rate': 0.021618284744896188, 'n_estimators': 2000, \n              'colsample_bytree': 0.78779835423704, 'colsample_bynode': 0.7824521219550881,\n              'verbose': -1, 'random_state': 42, 'reg_alpha': 0.191097277447348,\n              'reg_lambda': 10, 'extra_trees': True, 'num_leaves': 64, 'device': 'cpu', \n              'subsample': 0.6956746825486149}\nparams_7_0 = {'boosting_type': 'gbdt', 'objective': 'binary', 'metric': 'auc', 'max_depth': 2,\n              'learning_rate': 0.02072946855622066, 'n_estimators': 2000, \n              'colsample_bytree': 0.7662036532771843, 'colsample_bynode': 0.7322674198151792,\n              'verbose': -1, 'random_state': 42, 'reg_alpha': 0.1811123245593949, 'reg_lambda': 10,\n              'extra_trees': True, 'num_leaves': 64, 'device': 'cpu'}\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:46:45.602369Z","iopub.execute_input":"2024-05-27T06:46:45.602775Z","iopub.status.idle":"2024-05-27T06:46:45.624623Z","shell.execute_reply.started":"2024-05-27T06:46:45.602743Z","shell.execute_reply":"2024-05-27T06:46:45.623270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lets1 = {'num_leaves': 65, 'reg_alpha': 0.11342372425595186, 'subsample': 0.6062173391864109,\n         'scale_pos_weight': 1.0849474618579198, 'max_depth': 2, 'learning_rate': 0.021652251338798105,\n         'colsample_bytree': 0.5675417589766654,\n         'colsample_bynode': 0.5815526984656721}\nlets2 = {'num_leaves': 64, 'reg_alpha': 0.0902861437337476, \n         'subsample': 0.6381098887581637, 'scale_pos_weight': 0.9591670030599049,\n         'max_depth': 2, 'learning_rate': 0.0200257165852769, 'colsample_bytree': 0.6282679302366171, \n         'colsample_bynode': 0.5865906262748171}\nlets3 = {'boosting_type': 'gbdt', 'objective': 'binary', 'metric': 'auc', 'max_depth': 2, 'learning_rate': 0.022460532895681656, 'n_estimators': 2000,\n         'colsample_bytree': 0.7611228918657813, 'colsample_bynode': 0.710304338802183, \n         'verbose': -1, 'random_state': 42, 'reg_alpha': 0.19221596038351793, \n         'reg_lambda': 10, 'extra_trees': True, 'num_leaves': 64, 'device': 'cpu'}\nlets4 = {'boosting_type': 'gbdt', 'objective': 'binary', 'metric': 'auc', 'max_depth': 10,\n 'learning_rate': 0.028541149782907546, 'n_estimators': 2000, 'colsample_bytree': 0.7415159721237228, \n 'colsample_bynode': 0.7294712313122086, 'verbose': -1, 'random_state': 42,\n         'reg_alpha': 0.1752716267325066, 'reg_lambda': 10, 'extra_trees': True, \n         'num_leaves': 64, 'device': 'cpu'}\nlets5 = {'n_estimators': 2000, 'extra_trees': True, 'reg_lambda': 10, 'max_depth': 2,'metric': 'auc',\n         'device': 'cpu', 'num_leaves': 64,'subsample': 0.7,'verbose': -1,\n         'reg_alpha': 0.17383797340757948, 'learning_rate': 0.020608986305923917,\n         'colsample_bytree': 0.7258731414558194, 'colsample_bynode': 0.7002905698138404}\nlets6 = {'n_estimators': 2000, 'extra_trees': True, 'reg_lambda': 10, 'max_depth': 2,'metric': 'auc',\n         'verbose': -1, 'device': 'cpu', 'num_leaves': 64, 'subsample': 0.7,\n         'reg_alpha': 0.18417783212046715, 'learning_rate': 0.023727992400520988, \n         'colsample_bytree': 0.7406281320841832, 'colsample_bynode': 0.7214816209286798}\nlets7 = {'n_estimators': 2000, 'extra_trees': True, 'reg_lambda': 10, 'max_depth': 10, 'metric': 'auc',\n         'verbose': -1, 'device': 'cpu', 'num_leaves': 64, 'subsample': 0.7, \n         'reg_alpha': 0.18483615449923407, 'learning_rate': 0.020462388381424904,\n         'colsample_bytree': 0.7558586751984887, 'colsample_bynode': 0.7203592140287036}\nlets8 =  {'n_estimators': 2000, 'extra_trees': True, 'reg_lambda': 10,\n          'max_depth': 10,'metric': 'auc', 'verbose': -1, 'device': 'cpu',\n          'num_leaves': 64, 'subsample': 0.7, 'learning_rate': 0.02,\n          'reg_alpha': 0.1886404964061905, 'colsample_bytree': 0.731064157406144, \n          'colsample_bynode': 0.7258066058560992}\nlets9 = {'n_estimators': 2000, 'extra_trees': True, 'reg_lambda': 10, 'max_depth': 10, 'metric': 'auc',\n         'verbose': -1, 'device': 'cpu', 'num_leaves': 64, 'subsample': 0.7,\n         'learning_rate': 0.02, 'reg_alpha': 0.17407273959235478,\n         'colsample_bytree': 0.7227289693002997, 'colsample_bynode': 0.7363573384443878}\nlets10 = {'n_estimators': 2000, 'extra_trees': True, 'reg_lambda': 10, 'metric': 'auc',\n          'max_depth': 10, 'verbose': -1, 'device': 'cpu', 'num_leaves': 64,\n          'reg_alpha': 0.1872798150617034, 'learning_rate': 0.02173061619676849, \n          'colsample_bytree': 0.6880160185382109, 'colsample_bynode': 0.7236175666183192}","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:46:33.490101Z","iopub.execute_input":"2024-05-27T06:46:33.490820Z","iopub.status.idle":"2024-05-27T06:46:33.509073Z","shell.execute_reply.started":"2024-05-27T06:46:33.490782Z","shell.execute_reply":"2024-05-27T06:46:33.507670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\"\n# Function to filter numeric values\ndef filter_numeric(dictionary):\n    return {key: value for key, value in dictionary.items() if isinstance(value, (int, float)) and not isinstance(value, bool)}\n\n# Exclude non-numeric and boolean keys from each dictionary in parameter_list\nfiltered_parameter_list = [filter_numeric(params_dict) for params_dict in parameter_list]\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nrandom_state =  42\nimport logging\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n# Set LightGBM logging level to suppress warnings\nlogging.getLogger('lightgbm').setLevel(logging.ERROR)\nfitted_models_lgb = []\ncv_scores_lgb = []\n\nfor idx_train, idx_valid in cv.split(df_train, y, groups=weeks):\n    X_train, y_train = df_train.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = df_train.iloc[idx_valid], y.iloc[idx_valid]\n    \n    # Define LightGBM datasets\n    train_data = lgb.Dataset(X_train, label=y_train, categorical_feature=cat_cols)\n    valid_data = lgb.Dataset(X_valid, label=y_valid, categorical_feature=cat_cols)\n    \n    # Train LightGBM model\n    model = lgb.train(lets4, train_data, valid_sets=[valid_data],\n                      callbacks=[lgb.log_evaluation(200), lgb.early_stopping(100)])\n    \n    # Evaluate model on validation data\n    y_pred_valid = model.predict(X_valid)\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    \n    # Store model and score\n    fitted_models_lgb.append(model)\n    cv_scores_lgb.append(auc_score)\n\n# Display CV AUC scores\nprint(\"CV AUC scores: \", cv_scores_lgb)\nprint(\"Maximum CV AUC score: \", max(cv_scores_lgb))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-26T15:39:39.647807Z","iopub.execute_input":"2024-05-26T15:39:39.648351Z","iopub.status.idle":"2024-05-26T15:41:12.707541Z","shell.execute_reply.started":"2024-05-26T15:39:39.648313Z","shell.execute_reply":"2024-05-26T15:41:12.706212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Best_params: 0.7864786848788112, newparams = 0.7880536146298216, params = 0.7833253692767735,newest_params = 0.786943154667989, params_3_0 = 0.7871974034170311, params_4_0 = 0.788019955220965, params_5_0 = 0.7857632721753895, params_6_0 = 0.787092217764354, params_7_0= 0.7883850997009721","metadata":{}},{"cell_type":"markdown","source":"For New df_train with sin:\nlets3: 0.788331455018107, lets4 =  0.7890579873476686, lets5 = 0.7766798900058604, lets6 = 0.7762908533561736 lets9 = 0.7786880343806819, lets10 = 0.7830062059535079,best_params = 0.7844582187561046, params_7_0 = 0.7883186824745676, params_6_0 = 0.788044598716735, new_params = 0.788510420892876","metadata":{}},{"cell_type":"code","source":"\nweeks_test = df_test[\"WEEK_NUM\"]\ndf_test= df_test.drop(columns=[\"case_id\", \"WEEK_NUM\"])\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:02:29.511129Z","iopub.execute_input":"2024-05-27T07:02:29.511555Z","iopub.status.idle":"2024-05-27T07:02:29.533531Z","shell.execute_reply.started":"2024-05-27T07:02:29.511522Z","shell.execute_reply":"2024-05-27T07:02:29.532129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"auc_scores = {'lets3': 0.11129899728228725,\n 'lets4': 0.11140157129383867,\n 'params_3_0': 0.11103336559694384,\n 'lets10': 0.11054716266070069,\n 'best_params': 0.1107521621795469,\n 'params_7_0': 0.11129719401631374,\n 'params_6_0': 0.11125849805001183,\n 'new_params': 0.11132426422081006,\n 'params_4_0': 0.11108678469954703}\n\n# Calculate weights based on AUC scores\ntotal_auc = sum(auc_scores.values())\nweights = {param: auc_score / total_auc for param, auc_score in auc_scores.items()}","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:46:37.437674Z","iopub.execute_input":"2024-05-27T06:46:37.438114Z","iopub.status.idle":"2024-05-27T06:46:37.445401Z","shell.execute_reply.started":"2024-05-27T06:46:37.438079Z","shell.execute_reply":"2024-05-27T06:46:37.444124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights.values()","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:46:00.559747Z","iopub.execute_input":"2024-05-27T06:46:00.560185Z","iopub.status.idle":"2024-05-27T06:46:00.568000Z","shell.execute_reply.started":"2024-05-27T06:46:00.560152Z","shell.execute_reply":"2024-05-27T06:46:00.566681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import GroupKFold\n\nrandom_state = 42\nimport logging\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nlogging.getLogger('lightgbm').setLevel(logging.ERROR)\n\n# Supposing df_train, y, weeks, and best_params are defined\n# Somewhere else as context is not provided for these variables\n\n# Initialize cross-validation and performance tracking\ncv = GroupKFold(n_splits=5)\nfitted_models_lgb = []\ncv_scores_lgb = []\n\nfor idx_train, idx_valid in cv.split(df_train, y, groups=weeks):\n    X_train, y_train = df_train.iloc[idx_train], y.iloc[idx_train]\n    X_valid, y_valid = df_train.iloc[idx_valid], y.iloc[idx_valid]\n    \n    # Train LightGBM model using the scikit-learn API\n    model = lgb.LGBMClassifier(**best_params)\n    model.fit(X_train, y_train,\n              eval_set=[(X_valid, y_valid)],\n              eval_metric='auc',\n              early_stopping_rounds=100,\n              verbose=200,\n              categorical_feature=cat_cols)  # include cat_cols if it's defined earlier\n\n    # Predict probabilities on the validation set\n    y_pred_valid = model.predict_proba(X_valid)[:, 1]  # Get probabilities for the positive class\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    \n    # Store model and score\n    fitted_models_lgb.append(model)\n    cv_scores_lgb.append(auc_score)\n\n# Display CV AUC scores\nprint(\"CV AUC scores:\", cv_scores_lgb)\nprint(\"Maximum CV AUC score:\", max(cv_scores_lgb))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parameter_list = [lets3,lets4,params_3_0,lets10,best_params,params_7_0,params_6_0,new_params,params_4_0]","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:46:51.910614Z","iopub.execute_input":"2024-05-27T06:46:51.911752Z","iopub.status.idle":"2024-05-27T06:46:51.916878Z","shell.execute_reply.started":"2024-05-27T06:46:51.911713Z","shell.execute_reply":"2024-05-27T06:46:51.915347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\n\n%time\nmodel_combined = VotingClassifier([\n        ('lgb_0', lgb.LGBMClassifier(**parameter_list[0])),\n        ('lgb_1', lgb.LGBMClassifier(**parameter_list[1])),\n        ('lgb_2', lgb.LGBMClassifier(**parameter_list[2])), \n        ('lgb_3', lgb.LGBMClassifier(**parameter_list[3])), \n        ('lgb_4', lgb.LGBMClassifier(**parameter_list[4])), \n        ('lgb_5', lgb.LGBMClassifier(**parameter_list[5])), \n        ('lgb_6', lgb.LGBMClassifier(**parameter_list[6  ])), \n        ( 'lgb_7', lgb.LGBMClassifier(**parameter_list[7])),\n        ('lgb_8', lgb.LGBMClassifier(**parameter_list[8]))],weights= weights.values(), voting='soft')\nmodel_combined.fit(df_train, y)   \n\n# Predict using the VotingClassifier\n","metadata":{"execution":{"iopub.status.busy":"2024-05-27T06:46:53.280293Z","iopub.execute_input":"2024-05-27T06:46:53.280749Z","iopub.status.idle":"2024-05-27T07:00:13.298283Z","shell.execute_reply.started":"2024-05-27T06:46:53.280713Z","shell.execute_reply":"2024-05-27T07:00:13.296598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test1.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:05:25.628823Z","iopub.execute_input":"2024-05-27T07:05:25.630187Z","iopub.status.idle":"2024-05-27T07:05:25.638050Z","shell.execute_reply.started":"2024-05-27T07:05:25.630135Z","shell.execute_reply":"2024-05-27T07:05:25.636776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_combined.predict_proba(df_test1)[:, 1]\npredictions","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:05:28.524776Z","iopub.execute_input":"2024-05-27T07:05:28.525930Z","iopub.status.idle":"2024-05-27T07:05:29.773830Z","shell.execute_reply.started":"2024-05-27T07:05:28.525868Z","shell.execute_reply":"2024-05-27T07:05:29.772522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"array([0.00169359, 0.01471441, 0.00223605, 0.00603857, 0.03242051,\n       0.00794559, 0.02969321, 0.02066914, 0.01746453, 0.07338866])","metadata":{}},{"cell_type":"code","source":"#predictions = array([0.00384904, 0.02436812, 0.00464394, 0.01386842, 0.04836238,\n     #  0.00988114, 0.03007508, 0.01826774, 0.01931569, 0.06165025])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = predictions\ndf_subm.to_csv(\"submission.csv\")\ndf_subm","metadata":{"execution":{"iopub.status.busy":"2024-05-27T07:05:52.214108Z","iopub.execute_input":"2024-05-27T07:05:52.215545Z","iopub.status.idle":"2024-05-27T07:05:52.241048Z","shell.execute_reply.started":"2024-05-27T07:05:52.215484Z","shell.execute_reply":"2024-05-27T07:05:52.239591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering\n\nIn this part, we can see a simple example of joining tables via `case_id`. Here the loading and joining is done with polars library. Polars library is blazingly fast and has much smaller memory footprint than pandas. ","metadata":{}},{"cell_type":"markdown","source":"## Training LightGBM\n\nMinimal example of LightGBM training is shown below.","metadata":{}},{"cell_type":"markdown","source":"Evaluation with AUC and then comparison with the stability metric is shown below.","metadata":{}},{"cell_type":"code","source":"\"\"\"\ndef gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\nstability_score_train = gini_stability(df_train)\n#stability_score_valid = gini_stability(X_valid)\n#stability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}') \n#print(f'The stability score on the valid set is: {stability_score_valid}') \n#print(f'The stability score on the test set is: {stability_score_test}')  \n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission\n\nScoring the submission dataset is below, we need to take care of new categories. Then we save the score as a last step. ","metadata":{}},{"cell_type":"code","source":"\"\"\"\nX_submission = data_submission[cols_pred].to_pandas()\nX_submission = convert_strings(X_submission)\ncategorical_cols = X_train.select_dtypes(include=['category']).columns\n\nfor col in categorical_cols:\n    train_categories = set(X_train[col].cat.categories)\n    submission_categories = set(X_submission[col].cat.categories)\n    new_categories = submission_categories - train_categories\n    X_submission.loc[X_submission[col].isin(new_categories), col] = \"Unknown\"\n    new_dtype = pd.CategoricalDtype(categories=train_categories, ordered=True)\n    X_train[col] = X_train[col].astype(new_dtype)\n    X_submission[col] = X_submission[col].astype(new_dtype)\n\ny_submission_pred = gbm.predict(X_submission, num_iteration=gbm.best_iteration)\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nsubmission = pd.DataFrame({\n    \"case_id\": data_submission[\"case_id\"].to_numpy(),\n    \"score\": y_submission_pred\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Best of luck, and most importantly, enjoy the process of learning and discovery! \n\n<img src=\"https://i.imgur.com/obVWIBh.png\" alt=\"Image\" width=\"700\"/>","metadata":{}}]}