{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8171752,"sourceType":"datasetVersion","datasetId":4836407}],"dockerImageVersionId":30665,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🏠　Home Credit Baseline with some Encoding Idea\nWelcome to my notebook.\nI advanced baseline notebook with some Encoding Idea.\nHere, I will show you which encoding is so effective, \nAnd write my thinking.","metadata":{}},{"cell_type":"markdown","source":"## Importing package","metadata":{}},{"cell_type":"code","source":"import sys\nfrom pathlib import Path\nimport subprocess\nimport os\nimport gc\nfrom glob import glob\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom datetime import datetime\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nROOT = '/kaggle/input/home-credit-credit-risk-model-stability'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-26T07:56:46.299998Z","iopub.execute_input":"2024-05-26T07:56:46.300366Z","iopub.status.idle":"2024-05-26T07:56:46.306862Z","shell.execute_reply.started":"2024-05-26T07:56:46.300336Z","shell.execute_reply":"2024-05-26T07:56:46.305931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import TimeSeriesSplit, GroupKFold, StratifiedGroupKFold\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\n\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.impute import KNNImputer","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.308494Z","iopub.execute_input":"2024-05-26T07:56:46.309294Z","iopub.status.idle":"2024-05-26T07:56:46.318352Z","shell.execute_reply.started":"2024-05-26T07:56:46.309261Z","shell.execute_reply":"2024-05-26T07:56:46.317385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define method\n\nA description of each method is provided.\nIf you have any questions, feel free to comment！","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"\n        Method to set the data type according to the column name.\n        \n        Args:\n        - df (pl.DataFrame): Dataframe before data type setting.\n\n        Returns:\n        - df (pl.DataFrame): Dataframe after data type setting.\n        \"\"\"\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n\n    def handle_dates(df: pl.DataFrame) -> pl.DataFrame:\n        \"\"\"\n        Methods to preprocess dates to treat them as ints.\n        Args:\n        - df (pl.DataFrame): Dataframe before preprocessed.\n\n        Returns:\n        - df (pl.DataFrame): Data frame with an additional column \n        　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　　for how many days have passed since the base date.\n        \n        \"\"\"\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\n    def filter_cols(df):\n        \"\"\"\n        Methods to filter by percentage of missing values or number of unique values. \n        One of the methods to solve OOM error.\n        \"\"\"\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.9:\n                    df = df.drop(col)\n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n                if (freq == 1):\n                    df = df.drop(col)\n        \n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.325640Z","iopub.execute_input":"2024-05-26T07:56:46.325955Z","iopub.status.idle":"2024-05-26T07:56:46.341411Z","shell.execute_reply.started":"2024-05-26T07:56:46.325930Z","shell.execute_reply":"2024-05-26T07:56:46.340386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass Aggregator:\n    #Please add or subtract features yourself, be aware that too many features will take up too much space.\n    def num_expr(df):\n        \"\"\"\n        Extracting numerical features from the DataFrame (df).\n        This selects columns whose names end with \"P\" or \"A\", indicating some numerical measurements.\n        For each selected column, it creates an expression to compute the maximum value and aliases it accordingly.\n        \"\"\"\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        \n#         expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n#         expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n#         expr_var = [pl.var(col).alias(f\"var_{col}\") for col in cols]\n\n        return expr_max #+expr_last+expr_mean+expr_var\n    \n    def date_expr(df):\n        \"\"\"\n        Extracting date-related features from the DataFrame (df).\n        It selects columns whose names end with \"D\", representing date columns.\n        Similar to num_expr, it creates expressions to compute the maximum date value for each selected column and aliases them.\n        \"\"\"\n        cols = [col for col in df.columns if col[-1] in (\"D\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n#         expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n#         expr_mean = [pl.mean(col).alias(f\"mean_{col}\") for col in cols]\n        return  expr_max #+expr_last+expr_mean\n    \n    def str_expr(df):\n        \"\"\"\n        Extracting string-related features from the DataFrame (df).\n        It selects columns whose names end with \"M\", representing string columns.\n        Similar to num_expr, it creates expressions to compute the maximum date value for each selected column and aliases them.\n        \"\"\"\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n#         expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        #expr_count = [pl.count(col).alias(f\"count_{col}\") for col in cols]\n        return  expr_max #+expr_last+expr_count\n    \n    def other_expr(df):\n        \"\"\"\n        Extracting mixed datatype features from the DataFrame (df).\n        It selects columns whose names end with \"L\" and \"T\", representing bool and int columns.\n        Similar to num_expr, it creates expressions to compute the maximum date value for each selected column and aliases them.\n        \"\"\"\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n#         expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return  expr_max #+expr_last\n    \n    def count_expr(df):\n        \"\"\"\n        Extracting count features from the DataFrame (df).\n        It selects columns whose names end with \"num_group\", representing bool and int columns.\n        Similar to num_expr, it creates expressions to compute the maximum date value for each selected column and aliases them.\n        \"\"\"\n        cols = [col for col in df.columns if \"num_group\" in col]\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols] \n        #expr_min = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n#         expr_last = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        #expr_first = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return  expr_max #+expr_last\n    \n    def get_exprs(df):\n        \"\"\"\n        Aggregates all the expressions from the previous methods \n        to get a comprehensive list of feature extraction expressions.\n        \"\"\"\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.348992Z","iopub.execute_input":"2024-05-26T07:56:46.349279Z","iopub.status.idle":"2024-05-26T07:56:46.365426Z","shell.execute_reply.started":"2024-05-26T07:56:46.349255Z","shell.execute_reply":"2024-05-26T07:56:46.364402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    \"\"\"\n    Methods for reading single files\n    \"\"\"\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    if depth in [1,2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df)) \n    return df\n\ndef read_files(regex_path, depth=None):\n    \"\"\"\n    Methods for reading multiple files in once.\n    like bureau_b_*.  file \n    \"\"\"\n    chunks = []\n    \n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        chunks.append(df)\n    \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.419493Z","iopub.execute_input":"2024-05-26T07:56:46.419840Z","iopub.status.idle":"2024-05-26T07:56:46.428508Z","shell.execute_reply.started":"2024-05-26T07:56:46.419812Z","shell.execute_reply":"2024-05-26T07:56:46.427414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1, depth_2):\n    \"\"\"\n    A function that merges features created by other methods.\n    \n    \"\"\"\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.430915Z","iopub.execute_input":"2024-05-26T07:56:46.431247Z","iopub.status.idle":"2024-05-26T07:56:46.439552Z","shell.execute_reply.started":"2024-05-26T07:56:46.431216Z","shell.execute_reply":"2024-05-26T07:56:46.438699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_pandas(df_data, cat_cols=None):\n    \"\"\"\n    Converting　polars.DataFrame into　pandas.DataFrame\n    \"\"\"\n    \n    df_data = df_data.to_pandas()\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.440755Z","iopub.execute_input":"2024-05-26T07:56:46.441637Z","iopub.status.idle":"2024-05-26T07:56:46.453442Z","shell.execute_reply.started":"2024-05-26T07:56:46.441598Z","shell.execute_reply":"2024-05-26T07:56:46.452510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" \n    iterate through all the columns of a dataframe and modify the data type\n    to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.454709Z","iopub.execute_input":"2024-05-26T07:56:46.455553Z","iopub.status.idle":"2024-05-26T07:56:46.471817Z","shell.execute_reply.started":"2024-05-26T07:56:46.455520Z","shell.execute_reply":"2024-05-26T07:56:46.470931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Load and Preprocess","metadata":{}},{"cell_type":"markdown","source":"### Data load","metadata":{}},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.474843Z","iopub.execute_input":"2024-05-26T07:56:46.475246Z","iopub.status.idle":"2024-05-26T07:56:46.487920Z","shell.execute_reply.started":"2024-05-26T07:56:46.475212Z","shell.execute_reply":"2024-05-26T07:56:46.487046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_applprev_2.parquet\", 2),\n        read_file(TRAIN_DIR / \"train_person_2.parquet\", 2)\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:56:46.489486Z","iopub.execute_input":"2024-05-26T07:56:46.489827Z","iopub.status.idle":"2024-05-26T07:58:51.343369Z","shell.execute_reply.started":"2024-05-26T07:56:46.489797Z","shell.execute_reply":"2024-05-26T07:58:51.342304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Engineering with method","metadata":{}},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\nprint(\"train data shape:\\t\", df_train.shape)\ndel data_store\ndf_train = df_train.pipe(Pipeline.filter_cols)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:58:51.344686Z","iopub.execute_input":"2024-05-26T07:58:51.345042Z","iopub.status.idle":"2024-05-26T07:59:05.224856Z","shell.execute_reply.started":"2024-05-26T07:58:51.345005Z","shell.execute_reply":"2024-05-26T07:59:05.223909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### HandCrafted feature engineering","metadata":{}},{"cell_type":"markdown","source":"#### persontype \n\nI think the datatype of persontypes should be category.\nBut, before converting, the datatype of persontypes are int.","metadata":{}},{"cell_type":"code","source":"# Converting datatype \n\nperson_cols = df_train.select(\"^*persontype.*$\").columns\nfor col in person_cols:\n    df_train = df_train.with_columns(pl.col(col).cast(str))","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:59:05.226082Z","iopub.execute_input":"2024-05-26T07:59:05.226439Z","iopub.status.idle":"2024-05-26T07:59:05.542811Z","shell.execute_reply.started":"2024-05-26T07:59:05.226406Z","shell.execute_reply":"2024-05-26T07:59:05.541818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The percentage of total income from annuity is created as a feature fee.","metadata":{}},{"cell_type":"code","source":"cred_cols = [\"credamount_770A\"]\nannuity_cols = [\"annuity_780A\", \"monthsannuity_845L\"]\n\nfor cred_col in cred_cols:\n    for ann_col in annuity_cols:\n        df_train = df_train.with_columns((pl.col(ann_col)/pl.col(cred_col)).alias(cred_col+'_'+ann_col))","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:59:05.543959Z","iopub.execute_input":"2024-05-26T07:59:05.544239Z","iopub.status.idle":"2024-05-26T07:59:05.567363Z","shell.execute_reply.started":"2024-05-26T07:59:05.544216Z","shell.execute_reply":"2024-05-26T07:59:05.566565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"add Count Encoding of data(pl.String, pl.Boolean, pl.Categorical)\n\n※　I did this preprocess train and test data separately.\n  I cannot understand why this feature is so good._.\n","metadata":{}},{"cell_type":"code","source":"cnt_encoding_cols = df_train.select(pl.selectors.by_dtype([pl.String, pl.Boolean, pl.Categorical])).columns\n\nmappings = {}\nfor col in cnt_encoding_cols:\n    mappings[col] = df_train.group_by(col).len()\n\ndf_train_lazy = df_train.select(mappings.keys()).lazy()\n# df_train_lazy = pl.LazyFrame(df_train.select('case_id'))\n\nfor col, mapping in mappings.items():\n    remapping = {category: count for category, count in mapping.rows()}\n    remapping[None] = -2\n    expr = pl.col(col).replace(\n                remapping,\n                default=-1,\n            )\n    df_train_lazy = df_train_lazy.with_columns(expr.alias(col + '_cnt'))\n    del col, mapping, remapping\n    gc.collect()\n\ndel mappings\ntransformed_train = df_train_lazy.collect()\n\ndf_train = pl.concat([df_train, transformed_train.select(\"^*cnt$\")], how='horizontal')\ndel transformed_train, cnt_encoding_cols\n\ngc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-26T07:59:05.568480Z","iopub.execute_input":"2024-05-26T07:59:05.568760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature selecting with using corraration value.","metadata":{}},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ndf_train = reduce_mem_usage(df_train)\nprint(\"train data shape:\\t\", df_train.shape)\nnums=df_train.select_dtypes(exclude='category').columns\n\nnans_df = df_train[nums].isna()\nnans_groups={}\nfor col in nums:\n    cur_group = nans_df[col].sum()\n    try:\n        nans_groups[cur_group].append(col)\n    except:\n        nans_groups[cur_group]=[col]\ndel nans_df; x=gc.collect()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_group(grps):\n    use = []\n    for g in grps:\n        mx = 0; vx = g[0]\n        for gg in g:\n            n = df_train[gg].nunique()\n            if n>mx:\n                mx = n\n                vx = gg\n            #print(str(gg)+'-'+str(n),', ',end='')\n        use.append(vx)\n        #print()\n    print('Use these',use)\n    return use\n\ndef group_columns_by_correlation(matrix, threshold=0.8):\n    # 计算列之间的相关性\n    correlation_matrix = matrix.corr()\n\n    # 分组列\n    groups = []\n    remaining_cols = list(matrix.columns)\n    while remaining_cols:\n        col = remaining_cols.pop(0)\n        group = [col]\n        correlated_cols = [col]\n        for c in remaining_cols:\n            if correlation_matrix.loc[col, c] >= threshold:\n                group.append(c)\n                correlated_cols.append(c)\n        groups.append(group)\n        remaining_cols = [c for c in remaining_cols if c not in correlated_cols]\n    \n    return groups","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uses=[]\nfor k,v in nans_groups.items():\n    if len(v)>1:\n            Vs = nans_groups[k]\n            #cross_features=list(combinations(Vs, 2))\n            #make_corr(Vs)\n            grps= group_columns_by_correlation(df_train[Vs], threshold=0.8)\n            use=reduce_group(grps)\n            uses=uses+use\n            #make_corr(use)\n    else:\n        uses=uses+v\n    print('####### NAN count =',k)\n# print(uses)\n# print(len(uses))\nuses=uses+list(df_train.select_dtypes(include='category').columns)\n# print(len(uses))\ndf_train=df_train[uses]\ndf_train.drop(['requesttype_4525192L_cnt','max_empl_employedtotal_800L_cnt', 'max_empl_industry_691L_cnt'], axis=1, inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\ndevice='gpu'\n#n_samples=200000\nDRY_RUN = True if sample.shape[0] == 10 else False   \nif DRY_RUN:\n    device='cpu'\n    df_train = df_train.iloc[:10000]\nprint(device)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n        read_file(TEST_DIR / \"test_applprev_2.parquet\", 2),\n        read_file(TEST_DIR / \"test_person_2.parquet\", 2)\n    ]\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = feature_eng(**data_store)\nprint(\"test data shape:\\t\", df_test.shape)\ndel data_store\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in person_cols:\n   df_test = df_test.with_columns(pl.col(col).cast(str))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for cred_col in cred_cols:\n    for ann_col in annuity_cols:\n        df_test = df_test.with_columns((pl.col(cred_col)/pl.col(ann_col)).alias(cred_col+'_'+ann_col))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnt_encoding_cols = df_test.select(pl.selectors.by_dtype([pl.String, pl.Boolean, pl.Categorical])).columns\n\nmappings = {}\nfor col in cnt_encoding_cols:\n    mappings[col] = df_test.group_by(col).len()\n\ndf_test_lazy = df_test.select(mappings.keys()).lazy()\n# df_test_lazy = pl.LazyFrame(df_test.select('case_id'))\n\nfor col, mapping in mappings.items():\n    remapping = {category: count for category, count in mapping.rows()}\n    remapping[None] = -2\n    expr = pl.col(col).replace(\n                remapping,\n                default=-1,\n            )\n    df_test_lazy = df_test_lazy.with_columns(expr.alias(col + '_cnt'))\n    del col, mapping, remapping\ndel mappings\ntransformed_test = df_test_lazy.collect()\n\ndf_test = pl.concat([df_test, transformed_test.select(\"^*cnt$\")], how='horizontal')\ndel transformed_test, cnt_encoding_cols","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_test = df_test.select([col for col in df_train.columns if col not in ['requesttype_4525192L_cnt',\n                                                                         'max_empl_employedtotal_800L_cnt',\n                                                                         'max_empl_industry_691L_cnt',\n                                                                         \"target\"]])\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)\n\ndf_test, cat_cols = to_pandas(df_test, cat_cols)\ndf_test = reduce_mem_usage(df_test)\n\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Handle categorical features (Ordinal encoding)","metadata":{}},{"cell_type":"code","source":"# cat_list = [col for col in df_train.columns if df_train[col].dtype.name == 'category']\n\n# catfreq_dict = {}\n# catcatfreq_dict = {}\n\n# for col in cat_list:\n#     catfreq_dict[col] = len(list(df_train[col].value_counts()))\n#     catcatfreq_dict[col] = {}\n#     for d in dict(df_train[col].value_counts()).items():\n#         catcatfreq_dict[col][d[0]] = d[1]\n\n# catfreq_df = pd.DataFrame.from_dict(catfreq_dict, orient='index', columns=['Categories'])\n# display(catfreq_df.sort_values(by=\"Categories\", ascending=False).head())\n# display(catfreq_df.sort_values(by=\"Categories\", ascending=True).head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ordinal_enc = OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=np.nan)\n# df_train[cat_list] = ordinal_enc.fit_transform(df_train[cat_list])\n# df_test[cat_list] = ordinal_enc.transform(df_test[cat_list])\n# df_train[cat_list].head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"convert_cols = ['month_decision', 'weekday_decision']\n\ndf_train[convert_cols] = df_train[convert_cols].astype('category')\ndf_test[convert_cols] = df_test[convert_cols].astype('category')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yearmonth_cols = [\n 'max_dpdmaxdatemonth_442T',\n 'max_dpdmaxdatemonth_89T',\n 'max_dpdmaxdateyear_596T',\n 'max_dpdmaxdateyear_896T',\n 'max_overdueamountmaxdatemonth_284T',\n 'max_overdueamountmaxdatemonth_365T',\n 'max_overdueamountmaxdateyear_2T',\n 'max_overdueamountmaxdateyear_994T',\n 'max_pmts_month_158T',\n 'max_pmts_month_706T',\n 'max_pmts_year_1139T',\n 'max_pmts_year_507T',]\n\ndf_train[yearmonth_cols] = df_train[yearmonth_cols].astype(str).astype('category')\ndf_test[yearmonth_cols] = df_test[yearmonth_cols].astype(str).astype('category')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df_train[\"target\"]\nweeks = df_train[\"WEEK_NUM\"]\ndf_train= df_train.drop(columns=[\"target\", \"case_id\"])\n# df_train, y = SMOTE().fit_resample(df_train, y)\ncv = StratifiedGroupKFold(n_splits=5, shuffle=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cols = []\n\nfor c in df_train.columns:\n    if c not in df_test:\n        drop_cols.append(c)\n        \nif len(drop_cols) != 0:\n    df_train.drop(drop_cols, axis=1, inplace=True)\n    df_test.drop(drop_cols, axis=1, inplace=True)    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 10,  \n    \"learning_rate\": 0.05,\n    \"n_estimators\": 2000,  \n    \"colsample_bytree\": 0.8,\n    \"colsample_bynode\": 0.8,\n    \"verbose\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 0.1,\n    \"reg_lambda\": 10,\n    \"extra_trees\":True,\n    'num_leaves':64,\n    'categorical_feature ': 'auto',\n    \"device\": device, \n    \"verbose\": -1,\n}\n\nfitted_models = []\ncv_scores = []\n\n\nfor idx_train, idx_valid in cv.split(df_train, y, groups=weeks):#   Because it takes a long time to divide the data set, \n    X_train, y_train = df_train.iloc[idx_train], y.iloc[idx_train]# each time the data set is divided, two models are trained to each other twice, which saves time.\n    X_valid, y_valid = df_train.iloc[idx_valid], y.iloc[idx_valid]\n    model = lgb.LGBMClassifier(**params)\n    model.fit(\n        X_train, y_train,\n        eval_set = [(X_valid, y_valid)],\n        callbacks = [lgb.log_evaluation(200), lgb.early_stopping(100)] )\n    fitted_models.append(model)\n    y_pred_valid = model.predict_proba(X_valid)[:,1]\n    auc_score = roc_auc_score(y_valid, y_pred_valid)\n    cv_scores.append(auc_score)\n    \nprint(\"CV AUC scores: \", cv_scores)\nprint(\"Maximum CV AUC score: \", max(cv_scores))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VotingModel(BaseEstimator, RegressorMixin):\n    def __init__(self, estimators):\n        super().__init__()\n        self.estimators = estimators\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def predict(self, X):\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \n    def predict_proba(self, X):\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n\nmodel = VotingModel(fitted_models)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_importance(fitted_models[2], importance_type=\"split\", figsize=(10,50))\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = X_train.columns\nimportances = fitted_models[2].feature_importances_\nfeature_importance = pd.DataFrame({'importance':importances,'features':features}).sort_values('importance', ascending=False).reset_index(drop=True)\nfeature_importance\n\ndrop_list = []\nfor i, f in feature_importance.iterrows():\n    if f['importance']<80:\n        drop_list.append(f['features'])\nprint(f\"Number of features which are not important: {len(drop_list)} \")\n\nprint(drop_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"# df_test = df_test.drop(columns=[\"WEEK_NUM\"])\ndf_test = df_test.set_index(\"case_id\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pd.Series(model.predict_proba(df_test)[:, 1], index=df_test.index)\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred\ndf_subm.to_csv(\"submission.csv\")\ndf_subm","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}