{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30715,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport joblib\nimport lightgbm as lgb\nimport torch\nimport torch.nn as nn\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.preprocessing import LabelEncoder\nfrom catboost import CatBoostClassifier, Pool\n\nimport warnings\nwarnings.filterwarnings(action='ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-09T12:53:07.215198Z","iopub.execute_input":"2024-06-09T12:53:07.215574Z","iopub.status.idle":"2024-06-09T12:53:13.782953Z","shell.execute_reply.started":"2024-06-09T12:53:07.215538Z","shell.execute_reply":"2024-06-09T12:53:13.782108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:53:13.784995Z","iopub.execute_input":"2024-06-09T12:53:13.785614Z","iopub.status.idle":"2024-06-09T12:53:13.791236Z","shell.execute_reply.started":"2024-06-09T12:53:13.785577Z","shell.execute_reply":"2024-06-09T12:53:13.790224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" \n    Source: Guillame Martin, 2018\n    https://www.kaggle.com/code/gemartin/load-data-reduce-memory-usage\n    \n    \n    Iterate through all the columns of a dataframe and modify the data type\n    to reduce memory usage. Necessary to manage RAM due to very large dataset\n    \"\"\"\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type)==\"category\":\n            continue\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:53:13.792377Z","iopub.execute_input":"2024-06-09T12:53:13.792675Z","iopub.status.idle":"2024-06-09T12:53:13.808908Z","shell.execute_reply.started":"2024-06-09T12:53:13.792629Z","shell.execute_reply":"2024-06-09T12:53:13.807869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    \"\"\"\n    Source: Faruckan Saglam, 2024\n    https://www.kaggle.com/code/greysky/home-credit-baseline\n    \n    Edits by: Julie Anne Co, 2024\n    \n    Data Preparation methods for the Home Credit Risk Model Stability dataset\n    \"\"\"\n    \n    @staticmethod\n    def set_table_dtypes(df): \n        \"\"\"\n        Cast correct data types per column based on the column information embedded in the column names\n        Last character of column name indicates transformation and data types\n            * P - Transform DPD (Days past due)\n            * M - Masking categories\n            * A - Transform amount\n            * D - Transform date\n            * T - Unspecified Transform\n            * L - Unspecified Transform\n\n        \"\"\"\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        \"\"\"\n        Transform date columns into numeric columns by subtracting it from 'date_decision' column\n        \"\"\"\n        \n        for col in df.columns:\n            if (col[-1] in (\"D\",)) and ('count' not in col):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df): \n        \"\"\"\n        Remove Numeric columns with high perc_nulls (currently >0.95, modify to 0.97 to account for event rate)\n        Remove Categorical columns with only 1 unique value or >50 unique values to control complexity\n        Remove Numeric columns with no variance\n        Remove Month, Year, or num_group columns\n        \"\"\"\n        drop_cols = []\n        print(\"Columns dropped due to high nulls:\")\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n                if isnull > 0.95:\n                    print(col)\n                    drop_cols.append(col)\n                    df = df.drop(col)\n\n        print(\"Categorical Columns dropped either due to no variance or high granularity\")\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\", ]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 50):\n                    print(col)\n                    drop_cols.append(col)\n                    df = df.drop(col)\n                    \n            if (col[-1] not in [\"P\", \"A\", \"L\", \"M\"]) and (('month_' in col) or ('year_' in col)):\n                drop_cols.append(col)\n                df = df.drop(col)\n        \n        \n        print(\"Numeric Columns dropped due to no variance\")\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\", ]) & (df[col].dtype in[pl.Int64, pl.Float64]):\n                std = df[col].std()\n                if std == 0:\n                    print(col)\n                    drop_cols.append(col)\n                    df = df.drop(col)\n\n        return df, drop_cols\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:53:13.810934Z","iopub.execute_input":"2024-06-09T12:53:13.811261Z","iopub.status.idle":"2024-06-09T12:53:13.832854Z","shell.execute_reply.started":"2024-06-09T12:53:13.811235Z","shell.execute_reply":"2024-06-09T12:53:13.831676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    \"\"\"\n    Source: Faruckan Saglam, 2024\n    https://www.kaggle.com/code/greysky/home-credit-baseline\n    \n    Edits by: Julie Anne Co, 2024\n    \n    Aggregation methods for the Home Credit Risk Model Stability dataset\n    \"\"\"\n    \n    @staticmethod\n    def num_expr(df):\n        \"\"\"\n        Max, min for all columns\n        Mean, sum, std, sum for numeric columns\n        \"\"\"\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_2 = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n        \n        cols2 = [col for col in df.columns if col[-1] in (\"L\", \"A\")]\n        expr_3 = [pl.mean(col).alias(f\"mean_{col}\") for col in cols2] + [pl.std(col).alias(f\"std_{col}\") for col in cols2] + \\\n            [pl.sum(col).alias(f\"sum_{col}\") for col in cols2] \n        \n        return expr_1 + expr_2 + expr_3 \n    \n    @staticmethod\n    def applprev2_exprs(df):\n        \"\"\"\n        First values for applprev2 (previous applications) columns \n        \"\"\"\n        cols = [col for col in df.columns if \"num_group\" not in col]\n        expr_2 = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        return []\n    \n\n    @staticmethod\n    def bureau_a1(df):\n        \"\"\"\n        Min and Max for credit_bureau_a columns\n        Mean, sum, first, std for specific columns\n        \"\"\"\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        expr_2 = [pl.min(col).alias(f\"min_{col}\") for col in cols]\n\n        cols2 = [\n        'annualeffectiverate_199L', 'annualeffectiverate_63L',\n        'contractsum_5085717L', \n        'credlmt_230A', 'credlmt_935A',\n       'nominalrate_281L', 'nominalrate_498L',\n       'numberofcontrsvalue_258L', 'numberofcontrsvalue_358L',\n       'numberofinstls_229L', 'numberofinstls_320L',\n       'numberofoutstandinstls_520L', 'numberofoutstandinstls_59L',\n       'numberofoverdueinstlmax_1039L', 'numberofoverdueinstlmax_1151L',\n       'numberofoverdueinstls_725L', 'numberofoverdueinstls_834L'\n       ]\n        \n        \n        expr_3 = [pl.mean(col).alias(f\"mean_{col}\") for col in cols2] + [pl.std(col).alias(f\"std_{col}\") for col in cols2] + \\\n            [pl.sum(col).alias(f\"sum_{col}\") for col in cols2] +\\\n            [pl.first(col).alias(f\"first_{col}\") for col in cols2] \n\n        return expr_1 + expr_2 + expr_3    \n\n\n    @staticmethod\n    def bureau_b1(df):  \n        return []\n    \n    \n    @staticmethod\n    def bureau_b2(df):\n        \n        return []\n\n\n    @staticmethod\n    def deposit_exprs(df):\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] # + \\\n        return expr_1 \n\n    @staticmethod\n    def debitcard_exprs(df):\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] \n        return expr_1\n\n\n    @staticmethod\n    def person_expr(df):\n        cols1 = ['empl_employedtotal_800L', 'empl_employedfrom_271D', 'empl_industry_691L', \n                 'familystate_447L', 'incometype_1044T', 'sex_738L', 'housetype_905L', 'housingtype_772L',\n                 'isreference_387L', 'birth_259D', ]\n        expr_1 = [pl.first(col).alias(f\"first_{col}\") for col in cols1]\n        \n        expr_2 = [pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"), \n                  pl.col(\"mainoccupationinc_384A\").filter(pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")]\n        return expr_1 + expr_2 # + expr_4 # + expr_3\n    \n    @staticmethod\n    def person_2_expr(df):\n        cols = ['empls_economicalst_849M', 'empls_employedfrom_796D', 'empls_employer_name_740M'] # + \\\n        expr_1 = [pl.first(col).alias(f\"first_{col}\") for col in cols]\n        expr_2 = [pl.last(col).alias(f\"last_{col}\") for col in cols]\n        return expr_1 + expr_2\n\n    @staticmethod\n    def other_expr(df):\n        expr_1 = [pl.first(col).alias(f\"__other_{col}\") for col in df.columns if ('num_group' not in col) and (col != 'case_id')]\n        return expr_1 \n    \n    \n    @staticmethod\n    def tax_a_exprs(df):\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n        expr_1 = [pl.max(col).alias(f\"max_{col}\") for col in cols] + [pl.min(col).alias(f\"min_{col}\") for col in cols] + \\\n            [pl.last(col).alias(f\"last_{col}\") for col in cols] + \\\n            [pl.first(col).alias(f\"first_{col}\") for col in cols] + \\\n            [pl.mean(col).alias(f\"mean_{col}\") for col in cols] + \\\n            [pl.std(col).alias(f\"std_{col}\") for col in cols]\n        expr_4 = [pl.col(col).fill_null(strategy=\"zero\").apply(lambda x: x.max() - x.min()).alias(f\"max-min_gap_depth2_{col}\") for col in ['amount_4527230A']]\n\n        return expr_1 + expr_4\n\n\n    @staticmethod\n    def bureau_a2(df):\n        cols = [col for col in df.columns if (col[-1] in (\"T\",\"L\",\"M\",\"D\",\"P\",\"A\")) or (\"num_group\" in col)]\n\n        expr_1 = [pl.max(col).alias(f\"max_depth2_{col}\") for col in cols]\n        expr_2 = [pl.min(col).alias(f\"min_depth2_{col}\") for col in cols]\n        expr_3 = [pl.mean(col).alias(f\"mean_depth2_{col}\") for col in cols] + \\\n            [pl.std(col).alias(f\"std_{col}\") for col in cols]\n        \n        expr_4 = [pl.col(col).fill_null(strategy=\"zero\").apply(lambda x: x.max() - x.min()).alias(f\"max-min_gap_depth2_{col}\") for col in ['collater_valueofguarantee_1124L', 'pmts_dpd_1073P', 'pmts_overdue_1140A',]]\n\n        expr_ngc = [pl.count(\"num_group2\").alias(f\"count_depth2_a2_num_group2\")]\n\n        return expr_1 + expr_2 + expr_3 + expr_4 + expr_ngc\n    \n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df)\n\n        return exprs\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:53:13.834261Z","iopub.execute_input":"2024-06-09T12:53:13.834620Z","iopub.status.idle":"2024-06-09T12:53:13.869834Z","shell.execute_reply.started":"2024-06-09T12:53:13.834593Z","shell.execute_reply":"2024-06-09T12:53:13.868616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def agg_by_case(path, df):\n    path = str(path)\n    if '_applprev_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n    elif '_credit_bureau_a_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.bureau_a1(df))\n\n    elif '_credit_bureau_b_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.bureau_b1(df))\n\n    elif '_deposit_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.deposit_exprs(df))\n    elif '_debitcard_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.debitcard_exprs(df))\n        \n    elif '_tax_registry_a' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.tax_a_exprs(df))\n    elif '_tax_registry_b' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    elif '_tax_registry_c' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n    elif '_other_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.other_expr(df))\n    elif '_person_1' in path:\n        df = df.sort(\"num_group1\").group_by(\"case_id\").agg(Aggregator.person_expr(df))\n    elif '_person_2' in path:\n        df = df.group_by(\"case_id\").agg(Aggregator.person_2_expr(df))\n\n    elif '_credit_bureau_a_2' in path:\n        df = df.group_by(\"case_id\").agg(Aggregator.bureau_a2(df))\n    elif '_credit_bureau_b_2' in path:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\ndef read_file(path, depth=None): \n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = agg_by_case(path, df)\n    \n    print(f\"{path}: {df.shape}\")\n    \n    return df\n\ndef read_files(regex_path, depth=None):\n    \n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        if depth in [1, 2]:\n            df = agg_by_case(path, df)\n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    print(f\"{regex_path}: {df.shape}\")\n    \n    return df\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base.with_columns(\n            decision_month = pl.col(\"date_decision\").dt.month(),\n            decision_weekday = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(depth_0 + depth_1 + depth_2):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    print(df_data.info())\n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols\n\ndef pd_to_polars(df):\n    for col in df.columns:\n        if isinstance(df[col].dtype, pd.CategoricalDtype):\n            if pd.api.types.is_integer_dtype(df[col].cat.categories.dtype):\n                df[col] = df[col].astype(int)\n            elif pd.api.types.is_float_dtype(df[col].cat.categories.dtype):\n                df[col] = df[col].astype(float)\n\n    return pl.from_pandas(df)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:53:14.129973Z","iopub.execute_input":"2024-06-09T12:53:14.130338Z","iopub.status.idle":"2024-06-09T12:53:14.153869Z","shell.execute_reply.started":"2024-06-09T12:53:14.130287Z","shell.execute_reply":"2024-06-09T12:53:14.152731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n     \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n     \"depth_0\": [\n         read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n         read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n        \n     ],\n     \"depth_1\": [\n         read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n         read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n         read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n         read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n         read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n         read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n         read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n         read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n         read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n         read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n     ],\n     \"depth_2\": [\n         read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n         read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n         read_file(TRAIN_DIR / \"train_person_2.parquet\", 2),\n     ]\n }","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:53:15.415085Z","iopub.execute_input":"2024-06-09T12:53:15.416036Z","iopub.status.idle":"2024-06-09T12:56:49.104160Z","shell.execute_reply.started":"2024-06-09T12:53:15.416000Z","shell.execute_reply":"2024-06-09T12:56:49.102800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = feature_eng(**data_store)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:56:49.106311Z","iopub.execute_input":"2024-06-09T12:56:49.106630Z","iopub.status.idle":"2024-06-09T12:57:05.547575Z","shell.execute_reply.started":"2024-06-09T12:56:49.106603Z","shell.execute_reply":"2024-06-09T12:57:05.546482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, drop_col = df_train.pipe(Pipeline.filter_cols)\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:57:05.548853Z","iopub.execute_input":"2024-06-09T12:57:05.549133Z","iopub.status.idle":"2024-06-09T12:57:15.934392Z","shell.execute_reply.started":"2024-06-09T12:57:05.549110Z","shell.execute_reply":"2024-06-09T12:57:15.933301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:57:15.936471Z","iopub.execute_input":"2024-06-09T12:57:15.936839Z","iopub.status.idle":"2024-06-09T12:57:16.725056Z","shell.execute_reply.started":"2024-06-09T12:57:15.936811Z","shell.execute_reply":"2024-06-09T12:57:16.723949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def root_col(x):\n    \"\"\"\n    Get root information by removing prefixes\n    \"\"\"\n    split = x.split('_')\n    if split[0] in ['min', 'max', 'mean', 'var', 'mode', 'first', 'last', 'std', 'count']:\n        return '_'.join(split[1:])\n    return x","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:57:16.726408Z","iopub.execute_input":"2024-06-09T12:57:16.726790Z","iopub.status.idle":"2024-06-09T12:57:16.735885Z","shell.execute_reply.started":"2024-06-09T12:57:16.726763Z","shell.execute_reply":"2024-06-09T12:57:16.734868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = pd.read_csv(ROOT/'feature_definitions.csv')\ndrop_col = pd.DataFrame(drop_col, columns = ['Features'] )\ndrop_col['root_col'] = drop_col['Features'].apply(lambda x: root_col(x))\ndrop_col = drop_col.merge(features, left_on = 'root_col', right_on = 'Variable', how = 'left')\ndrop_col['main_info'] = drop_col['root_col'].apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:57:16.737437Z","iopub.execute_input":"2024-06-09T12:57:16.737771Z","iopub.status.idle":"2024-06-09T12:57:16.781675Z","shell.execute_reply.started":"2024-06-09T12:57:16.737744Z","shell.execute_reply":"2024-06-09T12:57:16.780738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_col.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:57:16.783176Z","iopub.execute_input":"2024-06-09T12:57:16.783526Z","iopub.status.idle":"2024-06-09T12:57:16.800534Z","shell.execute_reply.started":"2024-06-09T12:57:16.783497Z","shell.execute_reply":"2024-06-09T12:57:16.799540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[col for col in df_train.columns if 'last30dayturnover' in col]","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:52:17.157651Z","iopub.execute_input":"2024-06-09T12:52:17.158061Z","iopub.status.idle":"2024-06-09T12:52:17.166325Z","shell.execute_reply.started":"2024-06-09T12:52:17.158032Z","shell.execute_reply":"2024-06-09T12:52:17.165081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[col for col in df_train.columns if 'depth2_pmts_month' in col]","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:52:07.182435Z","iopub.execute_input":"2024-06-09T12:52:07.182855Z","iopub.status.idle":"2024-06-09T12:52:07.190739Z","shell.execute_reply.started":"2024-06-09T12:52:07.182824Z","shell.execute_reply":"2024-06-09T12:52:07.189671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"At a threshold of 95% null, 178 of 774 columns were dropped. Inspecting the dropped columns, multiple versions of columns exist for one type of information although there are versions retained in the training table.","metadata":{"execution":{"iopub.status.busy":"2024-06-09T12:45:33.896288Z","iopub.execute_input":"2024-06-09T12:45:33.897589Z","iopub.status.idle":"2024-06-09T12:45:33.904262Z","shell.execute_reply.started":"2024-06-09T12:45:33.897541Z","shell.execute_reply":"2024-06-09T12:45:33.902931Z"}}},{"cell_type":"markdown","source":"### Feature Selection","metadata":{}},{"cell_type":"code","source":"class Feature_Selector:\n    \"\"\"\n    Created by: Julie Anne Co, 2024\n    \n    Helper methods for feature selection. \n    Generation of stats, Data Type Transformers, Mann Whitney U Test, \n    Chi Square Test, Anova F-Value, Pearson Correlation\n    Correlation-based Feature Deletion\n    \"\"\"\n    @staticmethod\n    def stats_generator(df):\n        \"\"\"\n        Generate summary statistics\n        \"\"\"\n        stats = pd.DataFrame(df.describe(include = 'all')).T.reset_index()\n        stats = stats.rename(columns = {'index': 'columns'})\n        stats.insert(1, 'dtype', list(stats['columns'].apply(lambda x: df[x].dtype)))\n        stats['nunique'] = stats['columns'].apply(lambda x: df_train[x].nunique())\n        stats['perc_null'] = 1 - stats['count']/stats['count'].max()\n        stats = stats.drop(['unique', 'freq', 'top'], axis = 1)\n        for col in ['mean', 'std', 'min', '25%', '50%', '75%', 'max', 'perc_null']:\n            stats[col] = pd.to_numeric(stats[col], errors = 'coerce')\n        \n        return stats\n    \n    @staticmethod\n    def transform_category(df, cat_cols = None):\n        \"\"\"\n        Transform 'object' column types to 'category'\n        Useful for LightGBM categorical inputs\n        \"\"\"\n        if cat_cols is None:\n            cat_cols = list(df.select_dtypes(\"object\").columns)\n    \n        df[cat_cols] = df[cat_cols].astype(\"category\")\n        cat_cols = df.select_dtypes(\"category\").columns\n        return df, cat_cols\n    \n    @staticmethod\n    def cat_to_string(df, cat_cols = None):\n        \"\"\"\n        Transform 'category' columns to 'string'\n        Useful for CatBoost categorical inputs\n        \"\"\"\n        if cat_cols is None:\n            cat_cols = list(df.select_dtypes(\"category\").columns)\n    \n        df[cat_cols] = df[cat_cols].astype(\"object\")\n        cat_cols = df.select_dtypes(\"object\").columns\n        return df, cat_cols\n    \n    @staticmethod\n    def mann_whitney_test(df, col, target, alpha = 0.05):\n        \"\"\"\n        Checks if two samples (non-target vs target) come from the same distribution\n        Checking dependence of a column to target variable\n        Default alpha: 0.05\n        \"\"\"\n        from scipy.stats import mannwhitneyu\n        \n        if df[col].dtypes not in ['int64', 'float64'] and col != target:\n            return None\n        \n        sample1 = df[df[target] == 0][col]\n        sample2 = df[df[target] == 1][col]\n        \n        statistic, p_value = mannwhitneyu(sample1, sample2)\n        \n        return p_value\n        \n    @staticmethod\n    def chi_square(df, target_col, column):\n        \"\"\"\n        Returns the chi-square p-value for each column in the DataFrame\n        compared to the target column.\n\n        Requires target and subject column to be categorical/binary\n\n        \"\"\"\n        \n        from scipy.stats import chi2_contingency\n    \n        if df[column].dtypes != 'category' or column == target_col or df[column].nunique() > 50:\n            return None\n\n        contingency = pd.crosstab(df[column], df[target_col])\n        chi2, p_value, _, _ = chi2_contingency(contingency)\n\n        return p_value\n    \n    @staticmethod\n    def anova_f(df, target, col, impute = None):\n        from sklearn.feature_selection import f_classif\n        \n        if df[col].dtypes not in ['int64', 'float64']:\n            return None\n        \n        if col in ['case_id', 'target', 'WEEK_NUM', 'decision_month', 'decision_weekday']:\n            return None\n        \n        df = df[[col, target]]\n        \n        if impute is None:\n            df = df.dropna(subset=[col])\n            if len(df[target].unique()) < 2:\n                return None\n        \n        elif impute == 'mean':\n            df[col] = df.fillna(df[col].mean())\n        \n        elif impute == 'median':\n            df[col] = df.fillna(df[col].median())\n            \n        elif impute == 'zero':\n            df[col] = df.fillna(0)\n            \n        elif impute.isnumeric():\n            df[col] = df.fillna(eval(impute))\n        \n        f_statistic, p_val = f_classif(df[col].to_numpy().reshape(-1,1), df[target].to_numpy())\n        return p_val[0]\n    \n    @staticmethod\n    def pearson_corr(df, target, col):\n        \"\"\"\n        Returns correlation of 2 variables\n        \"\"\"\n        if target != col and df[col].dtype in ['int64', 'float64']:\n            return df[target].corr(df[col])\n        return None\n    \n    @staticmethod\n    def num_cols(df, exclude):\n        \"\"\"\n        Returns list of numeric columns\n        \"\"\"\n        num_cols = [col for col, dtype in df.dtypes.items() if col not in exclude and dtype in ['int64', 'float64']]\n        return num_cols\n    \n    @staticmethod\n    def correl_del(df, stats, threshold = 0.9):\n        \"\"\"\n        Returns columns to drop based on \n        \"\"\"\n        stats['corr'] = stats['corr'].apply(lambda x: np.abs(x))\n        stats = stats.sort_values(by = ['columns', 'corr', 'anova_dropna', 'perc_null', 'std'],\n                                  ascending = [True, False, True, True, False])\n        drop_cols = []\n        \n        for col in df.columns:\n            if col not in drop_cols:\n                cols = [x for x in df.columns if x != 'target']\n                corr = df[cols].corrwith(df[col])\n                high_corr = corr[np.abs(corr) > threshold].index\n                if high_corr > 1:\n                    drop = list(stats[stats['columns'].isin(high_corr)]['columns'][1:])\n                    print(f\"{col} -> Drop: {drop}\")\n                    drop_cols.append(drop)\n            else:\n                print(f\"{col}: Dropped\")\n                               \n        return drop_cols","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:45:37.643665Z","iopub.execute_input":"2024-06-02T16:45:37.644060Z","iopub.status.idle":"2024-06-02T16:45:37.677120Z","shell.execute_reply.started":"2024-06-02T16:45:37.644018Z","shell.execute_reply":"2024-06-02T16:45:37.675900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:45:37.697782Z","iopub.execute_input":"2024-06-02T16:45:37.698329Z","iopub.status.idle":"2024-06-02T16:45:45.794306Z","shell.execute_reply.started":"2024-06-02T16:45:37.698292Z","shell.execute_reply":"2024-06-02T16:45:45.793097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = Feature_Selector.transform_category(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:45:45.796038Z","iopub.execute_input":"2024-06-02T16:45:45.796392Z","iopub.status.idle":"2024-06-02T16:46:07.031504Z","shell.execute_reply.started":"2024-06-02T16:45:45.796357Z","shell.execute_reply":"2024-06-02T16:46:07.030313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:46:07.032909Z","iopub.execute_input":"2024-06-02T16:46:07.033257Z","iopub.status.idle":"2024-06-02T16:46:07.067895Z","shell.execute_reply.started":"2024-06-02T16:46:07.033229Z","shell.execute_reply":"2024-06-02T16:46:07.066812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(df_train.dtypes).to_csv('data_schema.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:48:15.539896Z","iopub.execute_input":"2024-06-02T16:48:15.540352Z","iopub.status.idle":"2024-06-02T16:48:15.558013Z","shell.execute_reply.started":"2024-06-02T16:48:15.540318Z","shell.execute_reply":"2024-06-02T16:48:15.556654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats = Feature_Selector.stats_generator(df_train)\nstats['chi_p'] = stats['columns'].apply(lambda x: Feature_Selector.chi_square(df_train, 'target', x))\nstats['anova_dropna'] = stats['columns'].apply(lambda x: Feature_Selector.anova_f(df_train, 'target', x))\nstats['corr'] = stats['columns'].apply(lambda x: Feature_Selector.pearson_corr(df_train, 'target', x))\nstats","metadata":{"execution":{"iopub.status.busy":"2024-06-02T14:23:20.321610Z","iopub.execute_input":"2024-06-02T14:23:20.322033Z","iopub.status.idle":"2024-06-02T14:25:22.884632Z","shell.execute_reply.started":"2024-06-02T14:23:20.321999Z","shell.execute_reply":"2024-06-02T14:25:22.883199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.to_csv('stats.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-02T14:25:22.886155Z","iopub.execute_input":"2024-06-02T14:25:22.886787Z","iopub.status.idle":"2024-06-02T14:25:22.915203Z","shell.execute_reply.started":"2024-06-02T14:25:22.886750Z","shell.execute_reply":"2024-06-02T14:25:22.914034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = Feature_Selector.num_cols(df_train, ['case_id', 'WEEK_NUM', 'target', 'decision_month', 'decision_weekday'])\nlen(num_cols)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T14:25:22.916925Z","iopub.execute_input":"2024-06-02T14:25:22.917479Z","iopub.status.idle":"2024-06-02T14:25:22.927213Z","shell.execute_reply.started":"2024-06-02T14:25:22.917447Z","shell.execute_reply":"2024-06-02T14:25:22.925710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Log of all columns deleted due to highly correlated variables","metadata":{}},{"cell_type":"code","source":"gc.collect()\nstats['abs_corr'] = stats['corr'].apply(lambda x: np.abs(x))\nstats = stats.sort_values(by = ['abs_corr', 'anova_dropna', 'perc_null', 'std'],\n                                  ascending = [False, True, True, False])\ndrop_cols = []\n        \nfor i, col in enumerate(num_cols):\n    if col not in drop_cols:\n        cols = [x for x in num_cols if x != 'target']\n        corr = df_train[cols].corrwith(df_train[col])\n        high_corr = corr[np.abs(corr) > 0.9].index\n        #if len(high_corr) > 1:\n        if len(high_corr) > 2:\n            drop = list(stats[stats['columns'].isin(high_corr)]['columns'][2:])\n            drop = [x for x in drop if x not in drop_cols]\n            drop_cols.extend(drop)\n            print(f\"{i}: {col} -> Drop: {drop}, {len(drop_cols)} cols to drop\")\n        else:\n            print(f\"{i}: {col} -> No High Corr, {len(drop_cols)} cols to drop\")\n    else:\n        print(f\"{i}: {col} -> Already Dropped, {len(drop_cols)} cols to drop\")\n\ndel cols\ndel corr\ndel high_corr\ndel drop\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-02T14:25:22.929177Z","iopub.execute_input":"2024-06-02T14:25:22.929621Z","iopub.status.idle":"2024-06-02T16:01:03.178122Z","shell.execute_reply.started":"2024-06-02T14:25:22.929576Z","shell.execute_reply":"2024-06-02T16:01:03.176652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nwith open('drop_num_cols_top2.pkl', 'wb') as f:\n    pickle.dump(drop_cols, f)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:02:45.138222Z","iopub.execute_input":"2024-06-02T16:02:45.139473Z","iopub.status.idle":"2024-06-02T16:02:45.144861Z","shell.execute_reply.started":"2024-06-02T16:02:45.139375Z","shell.execute_reply":"2024-06-02T16:02:45.143610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = reduce_mem_usage(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:01:03.188903Z","iopub.execute_input":"2024-06-02T16:01:03.189582Z","iopub.status.idle":"2024-06-02T16:01:20.236894Z","shell.execute_reply.started":"2024-06-02T16:01:03.189545Z","shell.execute_reply":"2024-06-02T16:01:20.235641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keep_cols = [x for x in df_train.columns if x not in drop_cols]\nlen(keep_cols)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:01:20.239042Z","iopub.execute_input":"2024-06-02T16:01:20.239532Z","iopub.status.idle":"2024-06-02T16:01:20.248856Z","shell.execute_reply.started":"2024-06-02T16:01:20.239493Z","shell.execute_reply":"2024-06-02T16:01:20.247619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(num_cols) - len(drop_cols)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:01:20.250446Z","iopub.execute_input":"2024-06-02T16:01:20.250849Z","iopub.status.idle":"2024-06-02T16:01:20.267238Z","shell.execute_reply.started":"2024-06-02T16:01:20.250817Z","shell.execute_reply":"2024-06-02T16:01:20.265939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Quick Check:\n- Nulls Capped at 95%\n- Categorical columns with uniqe values capped at 50\n","metadata":{}},{"cell_type":"code","source":"print(stats['perc_null'].quantile([0, 0.25, 0.5, 0.75, 1]))\nstats['perc_null'].hist(bins = 100);","metadata":{"execution":{"iopub.status.busy":"2024-06-02T05:59:33.086187Z","iopub.execute_input":"2024-06-02T05:59:33.086670Z","iopub.status.idle":"2024-06-02T05:59:33.525514Z","shell.execute_reply.started":"2024-06-02T05:59:33.086630Z","shell.execute_reply":"2024-06-02T05:59:33.524310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.to_csv('stats.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-02T06:00:10.234807Z","iopub.execute_input":"2024-06-02T06:00:10.235291Z","iopub.status.idle":"2024-06-02T06:00:10.258968Z","shell.execute_reply.started":"2024-06-02T06:00:10.235257Z","shell.execute_reply":"2024-06-02T06:00:10.257551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats[stats['dtype'] == 'object']['nunique'].quantile([0, 0.25, 0.5, 0.75, 1]))\nstats[stats['dtype'] == 'object']['nunique'].hist(bins = 100);","metadata":{"execution":{"iopub.status.busy":"2024-06-02T06:00:13.205429Z","iopub.execute_input":"2024-06-02T06:00:13.206474Z","iopub.status.idle":"2024-06-02T06:00:13.561099Z","shell.execute_reply.started":"2024-06-02T06:00:13.206428Z","shell.execute_reply":"2024-06-02T06:00:13.559924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stats['perc_null'].quantile([0, 0.25, 0.5, 0.75, 1]))\nstats['perc_null'].hist(bins = 100);","metadata":{"execution":{"iopub.status.busy":"2024-06-02T03:48:25.577424Z","iopub.execute_input":"2024-06-02T03:48:25.579555Z","iopub.status.idle":"2024-06-02T03:48:25.989763Z","shell.execute_reply.started":"2024-06-02T03:48:25.579501Z","shell.execute_reply":"2024-06-02T03:48:25.988635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train[list(stats['columns'])]","metadata":{"execution":{"iopub.status.busy":"2024-06-02T03:48:26.000952Z","iopub.execute_input":"2024-06-02T03:48:26.001391Z","iopub.status.idle":"2024-06-02T03:48:31.489184Z","shell.execute_reply.started":"2024-06-02T03:48:26.001352Z","shell.execute_reply":"2024-06-02T03:48:31.488011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_cols = ['case_id', 'WEEK_NUM', 'target', 'decision_month', 'decision_weekday']\nfeature_cols = [x for x in stats['columns'] if x not in id_cols]\nnum_cols = [x for x in stats[stats['dtype']!='object']['columns'] if x not in id_cols]","metadata":{"execution":{"iopub.status.busy":"2024-06-02T03:48:31.494188Z","iopub.execute_input":"2024-06-02T03:48:31.495219Z","iopub.status.idle":"2024-06-02T03:48:31.502827Z","shell.execute_reply.started":"2024-06-02T03:48:31.495179Z","shell.execute_reply":"2024-06-02T03:48:31.501672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop_cols = num_col_red1 + num_col_red2 + num_col_red3 + drop_cols_comb\nkeep_cols = [x for x in num_cols if x not in drop_cols]\nlen(keep_cols)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T05:13:46.926632Z","iopub.execute_input":"2024-06-02T05:13:46.927086Z","iopub.status.idle":"2024-06-02T05:13:46.936931Z","shell.execute_reply.started":"2024-06-02T05:13:46.927053Z","shell.execute_reply":"2024-06-02T05:13:46.935493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nwith open('drop_cols.pkl', 'wb') as f:\n    pickle.dump(drop_cols, f)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T05:14:38.847299Z","iopub.execute_input":"2024-06-02T05:14:38.848075Z","iopub.status.idle":"2024-06-02T05:14:38.854351Z","shell.execute_reply.started":"2024-06-02T05:14:38.848029Z","shell.execute_reply":"2024-06-02T05:14:38.852852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.merge(target[['case_id', 'target']], on = 'case_id', how = 'left')","metadata":{"execution":{"iopub.status.busy":"2024-06-02T05:22:42.681370Z","iopub.execute_input":"2024-06-02T05:22:42.681815Z","iopub.status.idle":"2024-06-02T05:22:56.633831Z","shell.execute_reply.started":"2024-06-02T05:22:42.681780Z","shell.execute_reply":"2024-06-02T05:22:56.632439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keep_cols = [x for x in df_train.columns if x not in drop_cols]","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:01:20.268782Z","iopub.execute_input":"2024-06-02T16:01:20.269168Z","iopub.status.idle":"2024-06-02T16:01:20.281217Z","shell.execute_reply.started":"2024-06-02T16:01:20.269127Z","shell.execute_reply":"2024-06-02T16:01:20.280322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train,cat_cols = Feature_Selector.cat_to_string(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:01:20.282569Z","iopub.execute_input":"2024-06-02T16:01:20.282880Z","iopub.status.idle":"2024-06-02T16:01:39.098144Z","shell.execute_reply.started":"2024-06-02T16:01:20.282854Z","shell.execute_reply":"2024-06-02T16:01:39.095973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pl.from_pandas(df_train[keep_cols]).write_csv('model_abt_pl_corr2_drop.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:01:39.102186Z","iopub.execute_input":"2024-06-02T16:01:39.103913Z","iopub.status.idle":"2024-06-02T16:02:27.839480Z","shell.execute_reply.started":"2024-06-02T16:01:39.103870Z","shell.execute_reply":"2024-06-02T16:02:27.837458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd_to_polars(df_train.drop(drop_cols, axis = 1)).write_csv('model_abt_pl_corr_drop.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-02T11:59:49.404065Z","iopub.execute_input":"2024-06-02T11:59:49.405013Z","iopub.status.idle":"2024-06-02T12:00:21.185431Z","shell.execute_reply.started":"2024-06-02T11:59:49.404977Z","shell.execute_reply":"2024-06-02T12:00:21.184036Z"},"trusted":true},"execution_count":null,"outputs":[]}]}