{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8426765,"sourceType":"datasetVersion","datasetId":4947825}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Install","metadata":{}},{"cell_type":"code","source":"!pip install scikit-learn-intelex","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:21:53.904023Z","iopub.execute_input":"2024-05-17T14:21:53.904420Z","iopub.status.idle":"2024-05-17T14:22:11.239905Z","shell.execute_reply.started":"2024-05-17T14:21:53.904391Z","shell.execute_reply":"2024-05-17T14:22:11.238397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"code","source":"from IPython.display import clear_output\n# suppress enjoying warnings from seaborn\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\n# for fiding file names\nimport os\nfrom pathlib import Path\nfrom glob import glob\nimport gc\n    \n# data processing libraries\nimport polars as pl\nimport numpy as np\nimport pandas as pd\n\n# for visualizing data\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# sklearn functions\nfrom sklearnex import patch_sklearn\npatch_sklearn()\nimport logging\nlogging.disable(logging.INFO)\n\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold\nfrom sklearn.preprocessing import LabelEncoder\n# for validating models\nfrom sklearn.metrics import roc_auc_score, roc_curve\n\n# LightGBM modeling\nimport lightgbm as lgb\n# Có thể thực hiện huấn luyện mô hình với các cách chọn feature khác nhau\n\n# define default colors for plots in notebook\nfrom matplotlib import cycler\nfrom matplotlib.colors import LinearSegmentedColormap\nCOLORS = [\"#068D9D\", \"#53599A\", \"#607BB0\", \"#6D9DC5\", \"#77BECF\", \"#80DED9\", \"#AEECEF\"]\nplt.rc('axes', facecolor='#E6E6E6', edgecolor='none', axisbelow=True, grid=True, prop_cycle=cycler('color', COLORS))\n\n# project CONSTANTS\nROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"\nSEED = 42\n\nplt.rcParams[\"font.size\"] = 8","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-17T14:22:52.969255Z","iopub.execute_input":"2024-05-17T14:22:52.969700Z","iopub.status.idle":"2024-05-17T14:22:56.999868Z","shell.execute_reply.started":"2024-05-17T14:22:52.969669Z","shell.execute_reply":"2024-05-17T14:22:56.998600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Info\n- Danh sách các ngưỡng đang có\n    - Số numerical feature\n    - Số categorical feature\n    - Số giá trị tương quan tối đa trên numerical feature\n    - Số feature sau cùng từ tree importance","metadata":{}},{"cell_type":"code","source":"# Lọc lấy pp fill null tốt nhất cho mỗi feature\n# Có thể là tiến hành sort rồi groupby lấy first :))))","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:30.771325Z","iopub.execute_input":"2024-05-16T01:14:30.771596Z","iopub.status.idle":"2024-05-16T01:14:30.775750Z","shell.execute_reply.started":"2024-05-16T01:14:30.771534Z","shell.execute_reply":"2024-05-16T01:14:30.774686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SUMMARY_PATH = Path(\"/kaggle/input/res-homecredit-feature-engineering-summary\")\ncategorical_summary = pl.read_parquet(SUMMARY_PATH / \"categorical_summary.parquet\")\nnumerical_summary = pl.read_parquet(SUMMARY_PATH / \"numerical_summary.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:23:00.604397Z","iopub.execute_input":"2024-05-17T14:23:00.604697Z","iopub.status.idle":"2024-05-17T14:23:00.756036Z","shell.execute_reply.started":"2024-05-17T14:23:00.604665Z","shell.execute_reply":"2024-05-17T14:23:00.755161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_summary","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:30.926220Z","iopub.execute_input":"2024-05-16T01:14:30.929695Z","iopub.status.idle":"2024-05-16T01:14:30.953308Z","shell.execute_reply.started":"2024-05-16T01:14:30.929658Z","shell.execute_reply":"2024-05-16T01:14:30.952527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_summary","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:30.954742Z","iopub.execute_input":"2024-05-16T01:14:30.955185Z","iopub.status.idle":"2024-05-16T01:14:30.965809Z","shell.execute_reply.started":"2024-05-16T01:14:30.955159Z","shell.execute_reply":"2024-05-16T01:14:30.964992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sort to get best fill_null_method\n# number: giảm dần\nnumerical_cols_sort = ['name', 'feature', 'agg', 'information_gain', 'std', 'fill_null_method']\nnumerical_descending_sort = [False, False, False, True, True, False]\nnumerical_summary = numerical_summary.sort(numerical_cols_sort, descending=numerical_descending_sort)\n\n# Category: tăng dần\ncategorical_cols_sort = ['name', 'feature', 'agg', 'p_value', 'chi2', 'nunique', 'fill_null_method']\ncategorical_descending_sort = [False, False, False, False, True, False, True] # null before mode\ncategorical_summary = categorical_summary.sort(categorical_cols_sort, descending=categorical_descending_sort)\n\n# lấy ra những fill_null_method tốt nhất\ncols_group_by = ['name', 'depth', 'feature', 'agg', 'null_percent']\n\nnumerical_exprs_group_by = [pl.first(\"fill_null_method\"), pl.first('information_gain'), pl.first('std')]\nnumerical_summary = numerical_summary.group_by(cols_group_by).agg(numerical_exprs_group_by)\n\ncategorical_exprs_group_by = [pl.first(\"fill_null_method\"), pl.first('chi2'), pl.first('p_value'), pl.first('nunique')]\ncategorical_summary = categorical_summary.group_by(cols_group_by).agg(categorical_exprs_group_by)\n\n# Sort again to ranking feature\nnumerical_cols_sort = ['information_gain', 'std']\nnumerical_descending_sort = [True, True]\nnumerical_summary = numerical_summary.sort(numerical_cols_sort, descending=numerical_descending_sort)\n\ncategorical_cols_sort = ['p_value', 'chi2', 'nunique']\ncategorical_descending_sort = [False, True, False] # null before mode\ncategorical_summary = categorical_summary.sort(categorical_cols_sort, descending=categorical_descending_sort)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:23:06.400747Z","iopub.execute_input":"2024-05-17T14:23:06.401090Z","iopub.status.idle":"2024-05-17T14:23:06.472280Z","shell.execute_reply.started":"2024-05-17T14:23:06.401026Z","shell.execute_reply":"2024-05-17T14:23:06.470741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_summary","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:31.022291Z","iopub.execute_input":"2024-05-16T01:14:31.022550Z","iopub.status.idle":"2024-05-16T01:14:31.031295Z","shell.execute_reply.started":"2024-05-16T01:14:31.022517Z","shell.execute_reply":"2024-05-16T01:14:31.030369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_summary","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:31.032284Z","iopub.execute_input":"2024-05-16T01:14:31.032466Z","iopub.status.idle":"2024-05-16T01:14:31.045140Z","shell.execute_reply.started":"2024-05-16T01:14:31.032444Z","shell.execute_reply":"2024-05-16T01:14:31.044062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def histogram_per_value(summary, metrics, hue):\n    unique_values = summary[hue].unique()\n    nunique = len(unique_values)\n    fig, axes = plt.subplots(nunique, len(metrics), figsize=(len(metrics) * 3, nunique * 3))\n    for r_idx, value in enumerate(unique_values):\n        for c_idx, metric in enumerate(metrics):\n            if value is None:\n                summary_filtered = summary.filter(pl.col(hue).is_null())\n            else:\n                summary_filtered = summary.filter(pl.col(hue) == value)\n            axes[r_idx][c_idx].hist(summary_filtered[metric])\n            axes[r_idx][c_idx].set_title(f\"{metric}: {value}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:31.049608Z","iopub.execute_input":"2024-05-16T01:14:31.049798Z","iopub.status.idle":"2024-05-16T01:14:31.057784Z","shell.execute_reply.started":"2024-05-16T01:14:31.049778Z","shell.execute_reply":"2024-05-16T01:14:31.056844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"histogram_per_value(numerical_summary, [\"information_gain\", \"std\"], \"agg\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"histogram_per_value(categorical_summary, [\"chi2\", \"p_value\", \"nunique\"], \"agg\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_min_max(df, metric):\n    min = df[metric].min()\n    max = df[metric].max()\n    print(f\"{metric}: {min:.4f} {max:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:46.824636Z","iopub.execute_input":"2024-05-16T01:14:46.825268Z","iopub.status.idle":"2024-05-16T01:14:46.829770Z","shell.execute_reply.started":"2024-05-16T01:14:46.825242Z","shell.execute_reply":"2024-05-16T01:14:46.828754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number\nprint(numerical_summary.shape)\n\nget_min_max(numerical_summary, \"information_gain\")\nget_min_max(numerical_summary, \"std\")\nget_min_max(numerical_summary, \"null_percent\")","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:46.830920Z","iopub.execute_input":"2024-05-16T01:14:46.831152Z","iopub.status.idle":"2024-05-16T01:14:46.845440Z","shell.execute_reply.started":"2024-05-16T01:14:46.831129Z","shell.execute_reply":"2024-05-16T01:14:46.844593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Categorical\nprint(categorical_summary.shape)\n\nget_min_max(categorical_summary, \"chi2\")\nget_min_max(categorical_summary, \"p_value\")\nget_min_max(categorical_summary, \"nunique\")\nget_min_max(categorical_summary, \"null_percent\")","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:46.846699Z","iopub.execute_input":"2024-05-16T01:14:46.846887Z","iopub.status.idle":"2024-05-16T01:14:46.855966Z","shell.execute_reply.started":"2024-05-16T01:14:46.846867Z","shell.execute_reply":"2024-05-16T01:14:46.854914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUMERICAL_FEATURE_AMOUNT = 1_389 # 1_389\nCATEGORICAL_FEATURE_AMOUNT = 378 # max 378\nfiltered_numerical_summary = numerical_summary.head(NUMERICAL_FEATURE_AMOUNT)\nfiltered_categorical_summary = categorical_summary.head(CATEGORICAL_FEATURE_AMOUNT)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:23:12.961531Z","iopub.execute_input":"2024-05-17T14:23:12.961793Z","iopub.status.idle":"2024-05-17T14:23:12.967396Z","shell.execute_reply.started":"2024-05-17T14:23:12.961765Z","shell.execute_reply":"2024-05-17T14:23:12.966157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_numerical_summary","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:46.866049Z","iopub.execute_input":"2024-05-16T01:14:46.866225Z","iopub.status.idle":"2024-05-16T01:14:46.878476Z","shell.execute_reply.started":"2024-05-16T01:14:46.866204Z","shell.execute_reply":"2024-05-16T01:14:46.877611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_categorical_summary","metadata":{"execution":{"iopub.status.busy":"2024-05-16T01:14:46.879776Z","iopub.execute_input":"2024-05-16T01:14:46.880043Z","iopub.status.idle":"2024-05-16T01:14:46.891960Z","shell.execute_reply.started":"2024-05-16T01:14:46.880012Z","shell.execute_reply":"2024-05-16T01:14:46.891025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Handle\n- Định hướng có đọc file test ở đây","metadata":{}},{"cell_type":"code","source":"class Pipeline:\n    def set_table_dtypes(df):\n        # Thiếu xử lý feature dạng T và L\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n    def under_sample(df, is_polars=True):\n        if is_polars:\n            df_1 = df.filter(df[\"target\"] == 1)\n            count_per_class = df_1.shape[0]\n            df_0 = df.filter(df[\"target\"] == 0).sample(n=count_per_class, seed=SEED, with_replacement=False)\n            df = pl.concat([df_0, df_1], how=\"vertical_relaxed\")\n        else:\n            df_1 = df[df[\"target\"] == 1]\n            count_per_class = df_1.shape[0]\n            df_0 = df[df[\"target\"] == 0].sample(n=count_per_class, random_state=SEED, replace=False)\n            df = pd.concat([df_0, df_1], ignore_index=True)\n        \n        return Pipeline.sort_df(df, 0, is_polars)\n    def sort_df(df, depth, is_polars=True):\n        # Sắp xếp để hàm first có ý nghĩa\n        if depth is None or depth == 0:\n            if is_polars:\n                return df.sort(\"case_id\")\n            else:\n                return df.sort_values(\"case_id\", ignore_index=True)\n        elif depth == 1:\n            cols = [\"case_id\", \"num_group1\"]\n            if is_polars:\n                return df.sort(cols)\n            else:\n                return df.sort_values(cols, ignore_index=True)\n        else:\n            cols = [\"case_id\", \"num_group1\", \"num_group2\"]\n            if is_polars:\n                return df.sort(cols)\n            else:\n                return df.sort_values(cols, ignore_index=True)\n    \n    def handle_dates(df):\n        # Lấy số ngày từ date_decision\n        # Hàm này được gọi khi mà đã thực hiện các phép tổng hợp xong\n        \n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                # Có thể cân nhắc thực hiện thêm các loại đặc trưng khác\n                # Chuyển các cột date thành số ngày so với date decision\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                # Hàm này thay đổi giá trị của cột col luôn chứ không thêm cột mới !\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\n    def filter_cols(df):\n        # Hàm này được áp dụng khi mà đã thực hiện các hàm tổng hợp !\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                # Bỏ các cột có tỉ lệ null trên ???\n                isnull = df[col].is_null().mean()\n                if isnull > 0.3:\n                    df = df.drop(col)\n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                # Bỏ các cột category có số giá trị unique trên 200\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n        \n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:23:16.664986Z","iopub.execute_input":"2024-05-17T14:23:16.665391Z","iopub.status.idle":"2024-05-17T14:23:16.692414Z","shell.execute_reply.started":"2024-05-17T14:23:16.665359Z","shell.execute_reply":"2024-05-17T14:23:16.691135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    def is_num_col(col, df):\n        # Chổ này cần xem xét kiểu dữ liệu để áp dụng phù hợp\n        return ((col[-1] in (\"P\", \"A\") or \n                ((col[-1] in (\"T\", \"L\"))) and df[col].dtype.is_numeric()))\n    def is_str_col(col, df):\n        return ((col[-1] == \"M\") or \n                ((col[-1] in (\"T\", \"L\")) and not df[col].dtype.is_numeric()))\n    def is_date_col(col, df):\n        return col[-1] == \"D\"\n    def agg_first(name, col):\n        return pl.first(col).alias(f\"{name}_first_{col}\")\n    def agg_last(name, col):\n        return pl.last(col).alias(f\"{name}_last_{col}\")\n    def agg_mean(name, col):\n        return pl.mean(col).alias(f\"{name}_mean_{col}\")\n    def agg_median(name, col):\n        return pl.median(col).alias(f\"{name}_median_{col}\")\n    def agg_max(name, col):\n        return pl.max(col).alias(f\"{name}_max_{col}\")\n    def agg_min(name, col):\n        return pl.min(col).alias(f\"{name}_min_{col}\")\n    def agg_std(name, col):\n        return pl.std(col).alias(f\"{name}_std_{col}\")\n    def agg_mode(name, col):\n        # Nếu null nhiều hơn thì vẫn lấy giá trị null luôn\n        return pl.col(col).drop_nulls().mode().first().alias(f\"{name}_mode_{col}\")\n    def agg_count(name, col):\n        return pl.count(col).alias(f\"{name}_count_{col}\")\n    def agg_nunique(name, col):\n        return pl.n_unique(col).alias(f\"{name}_nunique_{col}\")\n    def agg_sumvalue(name, col, value):\n        prefix = f\"{name}_sum{value}\"\n        return (pl.col(col) == value).sum().alias(f\"{prefix}_{col}\")\n    def get_exprs(name, features, aggs):\n        exprs = []\n        agg2expr = {\n            \"first\": Aggregator.agg_first,\n            \"last\": Aggregator.agg_last,\n            \"mean\": Aggregator.agg_mean,\n            \"median\": Aggregator.agg_median,\n            \"max\": Aggregator.agg_max,\n            \"min\": Aggregator.agg_min,\n            \"std\": Aggregator.agg_std,\n            \"mode\": Aggregator.agg_mode,\n            \"count\": Aggregator.agg_count,\n            \"nunique\": Aggregator.agg_nunique}\n\n        for feature, agg in zip(features, aggs):\n            if \"sum\" in agg:\n                value = agg[3:]\n                exprs.append(Aggregator.agg_sumvalue(name, feature, value))\n            else:\n                exprs.append(agg2expr[agg](name, feature))\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:23:22.118034Z","iopub.execute_input":"2024-05-17T14:23:22.118348Z","iopub.status.idle":"2024-05-17T14:23:22.137985Z","shell.execute_reply.started":"2024-05-17T14:23:22.118315Z","shell.execute_reply":"2024-05-17T14:23:22.136744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Get_data:\n    # Có những file chả cần đọc :)))\n    def get_base_df(features): # there no null val\n        feature2expr = {\n            \"month_decision\": pl.col(\"date_decision\").dt.month().alias(\"base_month_decision\"),\n            \"weekday_decision\": pl.col(\"date_decision\").dt.weekday().alias(\"base_weekday_decision\"),\n            \"year_decision\": pl.col(\"date_decision\").dt.year().alias(\"base_year_decision\")\n        }\n        train_df = pl.read_parquet(TRAIN_DIR / \"train_base.parquet\")\n        test_df = pl.read_parquet(TEST_DIR / \"test_base.parquet\")\n        \n        train_df = train_df.pipe(Pipeline.set_table_dtypes).pipe(Pipeline.under_sample) # có thực hiện undersample không ?\n        test_df = test_df.pipe(Pipeline.set_table_dtypes) # careful\n\n        train_df = train_df.with_columns([feature2expr[feature] for feature in features])\n        test_df = test_df.with_columns([feature2expr[feature] for feature in features])\n\n        return train_df, test_df\n    def get_df_0(name, paths, features):\n        # phải có case_id\n        print(features)\n        feature2name_feature = {feature: f\"{name}_{feature}\" for feature in features}\n        features = [\"case_id\"] + features.to_list()\n        # Train\n        chunks = []\n        for path in paths:\n            df = pl.read_parquet(path, columns=features)\n            df = df.pipe(Pipeline.set_table_dtypes)\n            df = Pipeline.sort_df(df, 0)\n            chunks.append(df)\n        df = pl.concat(chunks, how=\"vertical_relaxed\")\n        return df.rename(feature2name_feature)\n        # relaxed: chuyển đổi để cùng kiểu dữ liệu giữa các hàng\n    def get_train_df_1_2(depth, paths, exprs):\n        chunks = []\n        schema = None\n        for path in paths:\n            df = pl.read_parquet(path)\n            df = df.pipe(Pipeline.set_table_dtypes)\n            schema = df.schema\n            df = Pipeline.sort_df(df, depth)\n            df = df.group_by(\"case_id\").agg(exprs)\n            chunks.append(df)\n        return pl.concat(chunks, how=\"vertical_relaxed\"), schema\n    def get_test_df_1_2(depth, paths, schema, exprs):\n        chunks = []\n        type_exprs = [pl.col(col).cast(type) for col, type in schema.items()]\n        for path in paths:\n            df = pl.read_parquet(path)\n            df = df.with_columns(type_exprs)\n            df = Pipeline.sort_df(df, depth)\n            df = df.group_by(\"case_id\").agg(exprs)\n            chunks.append(df)\n        return pl.concat(chunks, how=\"vertical_relaxed\")\n    def get_df_1_2(depth, train_paths, test_paths, exprs):\n        # gán dtype của train cho test\n        # Train\n        train_df, schema = Get_data.get_train_df_1_2(depth, train_paths, exprs)\n        test_df = Get_data.get_test_df_1_2(depth, test_paths, schema, exprs)\n        # relaxed: chuyển đổi để cùng kiểu dữ liệu giữa các hàng\n        return train_df, test_df\n    def get_data(summary):\n        # Ủa rồi lọc feature đâu ?\n        \n        # name | depth | feature | agg | fill_null_method\n        # get base\n        print(\"base\")\n        train_df_base, test_df_base = Get_data.get_base_df(summary.filter(pl.col(\"name\") == \"base\")[\"feature\"])\n        \n        for depth, name in summary[[\"depth\", \"name\"]].unique().rows():\n            if name == \"base\":\n                continue\n            else:\n                print(name)\n                _summary = summary.filter(pl.col(\"name\") == name)\n                \n                train_paths = filter(lambda _: name in _.name ,TRAIN_DIR.iterdir())\n                test_paths = filter(lambda _: name in _.name ,TEST_DIR.iterdir())\n\n                features = _summary[\"feature\"]\n                print(features)\n                if depth == 1 or depth == 2:\n                    \n                    aggs = _summary[\"agg\"]\n                    exprs = Aggregator.get_exprs(name, features, aggs)\n                    \n#                     train_df = Get_data.get_df_1_2(depth, train_paths, exprs)\n#                     test_df = Get_data.get_df_1_2(depth, test_paths, exprs)\n                    train_df, test_df = Get_data.get_df_1_2(depth, train_paths, test_paths, exprs)\n                else:\n                    train_df = Get_data.get_df_0(name, train_paths, features)\n                    test_df = Get_data.get_df_0(name, test_paths, features)\n\n                # relaxed: chuyển đổi để cùng kiểu dữ liệu giữa các hàng\n\n                # Thực hiện join table\n                train_df_base = train_df_base.join(train_df, how=\"left\", on=\"case_id\", suffix=f\"{name}_\")\n\n                test_df_base = test_df_base.join(test_df, how=\"left\", on=\"case_id\", suffix=f\"{name}_\")\n                # Thực hiện feature eng: đổi các kiểu dữ liệu date\n        train_df_base = train_df_base.pipe(Pipeline.handle_dates)\n        test_df_base = test_df_base.pipe(Pipeline.handle_dates)\n\n        return train_df_base, test_df_base","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:23:25.335185Z","iopub.execute_input":"2024-05-17T14:23:25.335561Z","iopub.status.idle":"2024-05-17T14:23:25.364108Z","shell.execute_reply.started":"2024-05-17T14:23:25.335518Z","shell.execute_reply":"2024-05-17T14:23:25.362990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Fill_null:\n    def get_zero(sr:pl.Series):\n        return 0\n    def get_median(sr:pl.Series):\n        return sr.median()\n    def get_mean(sr:pl.Series):\n        return sr.mean()\n    def get_max(sr:pl.Series):\n        return sr.max()\n    def get_min(sr:pl.Series):\n        return sr.min()\n    def get_mode(sr:pl.Series):\n        return sr.mode()[0]\n    def get_null(sr:pl.Series):\n        return \"null\"\n    def get_value(sr:pl.Series, method):\n        method2value = {\n            \"zero\": Fill_null.get_zero,\n            \"median\": Fill_null.get_median,\n            \"mean\": Fill_null.get_mean,\n            \"max\": Fill_null.get_max,\n            \"min\": Fill_null.get_min,\n            \"mode\": Fill_null.get_mode,\n            \"null\": Fill_null.get_null\n        }\n        return method2value[method](sr.drop_nulls())\n    def fill(summary, train_df, test_df):\n        columns = train_df.columns\n        exprs = []\n        feature2null_value = {}\n        for name, feature, agg, method, type in summary[[\"name\", \"feature\", \"agg\", \"fill_null_method\", \"type\"]].rows(): # name, depth, feature, agg, fill_null_method\n            if agg:\n                col_name = f\"{name}_{agg}_{feature}\"\n            else:\n                col_name = f\"{name}_{feature}\"\n            \n            if method: \n                value = Fill_null.get_value(train_df[col_name], method)\n                print(col_name, method, value)\n\n            else: # Vẫn có trường hợp tập train không có null nhưng tập test thì có\n                if type == \"categorical\":\n                    value = Fill_null.get_value(train_df[col_name], \"null\") # cân nhăc\n                else:\n                    value = Fill_null.get_value(train_df[col_name], \"mean\") # cân nhắc\n                \n            feature2null_value[col_name] = value\n            exprs.append(pl.col(col_name).fill_null(value))\n        train_df = train_df.with_columns(exprs) # Thừa nhưng không sao\n        test_df = test_df.with_columns(exprs)\n        return train_df, test_df, feature2null_value","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:23:28.712609Z","iopub.execute_input":"2024-05-17T14:23:28.712919Z","iopub.status.idle":"2024-05-17T14:23:28.732949Z","shell.execute_reply.started":"2024-05-17T14:23:28.712885Z","shell.execute_reply":"2024-05-17T14:23:28.731166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Đọc cái file đó\n- Group by file, feature, agg, chọn pp fill null tốt nhất\n- fill null: mỗi đặc trưng chỉ nên dùng 1 pp fill null\n- Lượt bỏ feature bằng ngưỡng\n- Xử lý dữ liệu số: Bỏ đặc trưng bằng ma trận hệ số tương quan\n- Trộn đặc trưng số, categorical lại rồi tính feature importance bằng pp cây -> Chọn tiếp\n- Thử tách dữ liệu ra rồi test độ chính xác\n- Có thể cần tiếp vài pp xử lý đặc trưng như quantilecut","metadata":{}},{"cell_type":"code","source":"%%time\ncols = [\"name\", \"depth\", \"feature\", \"agg\", \"fill_null_method\"]\nfiltered_summary = pl.concat([filtered_numerical_summary[cols], filtered_categorical_summary[cols]])\ntype_sr = pl.Series([\"numerical\"] * filtered_numerical_summary.shape[0] + [\"categorical\"] * filtered_categorical_summary.shape[0])\nfiltered_summary = filtered_summary.with_columns(type_sr.alias(\"type\"))\ntrain_df, test_df = Get_data.get_data(filtered_summary)\nclear_output()\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:23:32.377926Z","iopub.execute_input":"2024-05-17T14:23:32.378708Z","iopub.status.idle":"2024-05-17T14:32:45.645613Z","shell.execute_reply.started":"2024-05-17T14:23:32.378665Z","shell.execute_reply":"2024-05-17T14:32:45.643759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:16.750797Z","iopub.execute_input":"2024-05-17T14:33:16.751252Z","iopub.status.idle":"2024-05-17T14:33:16.850895Z","shell.execute_reply.started":"2024-05-17T14:33:16.751213Z","shell.execute_reply":"2024-05-17T14:33:16.849648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:20.460218Z","iopub.execute_input":"2024-05-17T14:33:20.460487Z","iopub.status.idle":"2024-05-17T14:33:20.491899Z","shell.execute_reply.started":"2024-05-17T14:33:20.460460Z","shell.execute_reply":"2024-05-17T14:33:20.490557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df, test_df, feature2null_value = Fill_null.fill(filtered_summary, train_df, test_df)\ntrain_df = train_df.to_pandas()\ntest_df = test_df.to_pandas()\nclear_output()\nprint(\"Done\")\nprint((test_df.isnull().sum() > 0).all())","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:23.034432Z","iopub.execute_input":"2024-05-17T14:33:23.034719Z","iopub.status.idle":"2024-05-17T14:33:28.261982Z","shell.execute_reply.started":"2024-05-17T14:33:23.034689Z","shell.execute_reply":"2024-05-17T14:33:28.261081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_feature = lambda info: (f\"{info['name']}_{info['agg']}_{info['feature']}\"\n                if info[\"agg\"] else f\"{info['name']}_{info['feature']}\")\n\nnumerical_features = [get_feature(info) for info in filtered_numerical_summary.rows(named=True)]\ncategorical_features = [get_feature(info) for info in filtered_categorical_summary.rows(named=True)]\n\ntrain_numerical_df = train_df[numerical_features]","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:37.030329Z","iopub.execute_input":"2024-05-17T14:33:37.030606Z","iopub.status.idle":"2024-05-17T14:33:37.526248Z","shell.execute_reply.started":"2024-05-17T14:33:37.030573Z","shell.execute_reply":"2024-05-17T14:33:37.524974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features[:10]","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:39.177300Z","iopub.execute_input":"2024-05-17T14:33:39.177605Z","iopub.status.idle":"2024-05-17T14:33:39.187498Z","shell.execute_reply.started":"2024-05-17T14:33:39.177568Z","shell.execute_reply":"2024-05-17T14:33:39.186111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_numerical_df","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:42.225638Z","iopub.execute_input":"2024-05-17T14:33:42.226236Z","iopub.status.idle":"2024-05-17T14:33:42.305614Z","shell.execute_reply.started":"2024-05-17T14:33:42.226197Z","shell.execute_reply":"2024-05-17T14:33:42.304537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Remove outliers","metadata":{}},{"cell_type":"markdown","source":"## Numerical","metadata":{}},{"cell_type":"code","source":"def clip_numerical_outliers(train_df: pl.DataFrame, test_df: pl.DataFrame, columns):\n    print(\"Outlier count\")\n    for col in columns:\n        train_sr = train_df[col]\n        test_sr = test_df[col]\n        Q1 = train_sr.quantile(0.25)\n        Q3 = train_sr.quantile(0.75)\n        IQR = Q3 - Q1\n        low_fence = Q1 - 1.5 * IQR\n        up_fence = Q1 + 1.5 * IQR\n\n        train_ouliers_count = ((train_sr < low_fence) | (train_sr > up_fence)).sum()\n        test_ouliers_count = ((test_sr < low_fence) | (test_sr > up_fence)).sum()\n        print(f\"\\t{col}: Train-{train_ouliers_count} | Test-{test_ouliers_count}\")\n\n        train_df[col] = train_sr.clip(lower=low_fence, upper=up_fence)\n        test_df[col] = test_sr.clip(lower=low_fence, upper=up_fence)\n    return train_df, test_df","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:45.379225Z","iopub.execute_input":"2024-05-17T14:33:45.379547Z","iopub.status.idle":"2024-05-17T14:33:45.390111Z","shell.execute_reply.started":"2024-05-17T14:33:45.379499Z","shell.execute_reply":"2024-05-17T14:33:45.388743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# _train_df, _test_df = clip_numerical_outliers(train_df, test_df, numerical_features)\n# clear_output()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical","metadata":{}},{"cell_type":"code","source":"def clip_categorical_outliers(train_df: pl.DataFrame, test_df: pl.DataFrame, columns, thres, null_values):\n    # Dùng chung cho cả tập test\n    print(\"Outlier count\")\n    for col, null_value in zip(columns, null_values):\n        train_sr = train_df[col]\n        test_sr = test_df[col]\n\n        value_counts = train_sr.value_counts() / len(train_sr) # Không có null trong này nhưng sẽ có \"null\"\n        rare_values = value_counts[value_counts < thres].index\n        \n        train_ouliers_count = train_sr.isin(rare_values).sum()\n        test_ouliers_count = test_sr.isin(rare_values).sum()\n\n        clip = lambda value: null_value if value in rare_values else value\n        train_df[col] = train_sr.apply(clip)\n        test_df[col] = test_sr.apply(clip)\n        print(f\"\\t{col}: Train-{train_ouliers_count} | Test-{test_ouliers_count}\")\n    return train_df, test_df","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:48.248095Z","iopub.execute_input":"2024-05-17T14:33:48.248367Z","iopub.status.idle":"2024-05-17T14:33:48.257255Z","shell.execute_reply.started":"2024-05-17T14:33:48.248340Z","shell.execute_reply":"2024-05-17T14:33:48.256150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# null_values = [feature2null_value[feature] for feature in categorical_features]\n# thres = 0.01\n# _train_df, _test_df = clip_categorical_outliers(train_df, test_df, categorical_features, thres, null_values)\n# clear_output()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Select numerical features by corr","metadata":{}},{"cell_type":"code","source":"def getNumericalFeaturesByCorr(corr_mtrx, numerical_features, numerical_feature2rank, thres, verbose=False):\n    # Lập list corr sort theo abs(corr): với 2 feature đằng sau\n    # Trên chọn feature rank thấp hơn ,bỏ feature rank cao hơn: rank từ thấp tới cao\n    # List đánh dấu: feature còn được chọn không\n    # Duyệt tới corr kết tiếp: nếu có 1 trong 2 feature bỏ bị bỏ qua thì bỏ qua corr đó\n    is_selected = [True] * len(numerical_features)\n    corrs = []\n    for r_idx in range(1, len(is_selected)):\n        for c_idx in range(r_idx):\n            corr = corr_mtrx.item(r_idx, c_idx)\n            corrs.append([corr, (r_idx, c_idx)])\n    corrs.sort(key=lambda _: abs(_[0]), reverse=True) # Giảm dần\n    for corr, idxs in corrs:\n        if abs(corr) < thres:\n            break\n        elif (not is_selected[idxs[0]]) or (not is_selected[idxs[1]]):\n            continue\n        else:\n            feature0 = numerical_features[idxs[0]]\n            feature1 = numerical_features[idxs[1]]\n\n            if numerical_feature2rank[feature0] > numerical_feature2rank[feature1]: # Rank decrease\n                is_selected[idxs[0]] = False\n            else:\n                is_selected[idxs[1]] = False\n\n    seleted_numerical_features = [feature for is_selected, feature in zip(is_selected, numerical_features) if is_selected]\n    if verbose:\n        print(f\"Còn lại: {len(seleted_numerical_features)} / {train_numerical_df.shape[1]} numerical feature\")\n    return seleted_numerical_features","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:51.788190Z","iopub.execute_input":"2024-05-17T14:33:51.788473Z","iopub.status.idle":"2024-05-17T14:33:51.800385Z","shell.execute_reply.started":"2024-05-17T14:33:51.788443Z","shell.execute_reply":"2024-05-17T14:33:51.799413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kiểm tra\ndef checkCorr(corr_mtrx):\n    max = 0\n    for r_idx in range(1, corr_mtrx.shape[0]):\n        for c_idx in range(r_idx):\n            corr = corr_mtrx.item(r_idx, c_idx)\n            if abs(corr) > max:\n                max = abs(corr)\n    print(max)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:54.685244Z","iopub.execute_input":"2024-05-17T14:33:54.687341Z","iopub.status.idle":"2024-05-17T14:33:54.693961Z","shell.execute_reply.started":"2024-05-17T14:33:54.687291Z","shell.execute_reply":"2024-05-17T14:33:54.692667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# features = categorical_features + seleted_numerical_features\n# types = [\"categorical\"] * len(categorical_features) + [\"numerical\"] * len(seleted_numerical_features)\n# train_df = train_df[[\"case_id\", \"target\", \"WEEK_NUM\"] + features]\n# test_df = test_df[[\"case_id\", \"WEEK_NUM\"] + features]","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:56.821012Z","iopub.execute_input":"2024-05-17T14:33:56.821357Z","iopub.status.idle":"2024-05-17T14:33:56.827020Z","shell.execute_reply.started":"2024-05-17T14:33:56.821324Z","shell.execute_reply":"2024-05-17T14:33:56.825795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Custom label encoder","metadata":{}},{"cell_type":"code","source":"class CustomLabelEncoder:\n    def fit(self, y: pd.Series, value_fill_null): # Lấy giá trị dùng để fill null nhét vào\n         # Nếu giá trị đó không tồn tại ???\n        self.classes_ = y.unique()\n        self.class_to_index_ = {cls: idx for idx, cls in enumerate(self.classes_)}\n        if value_fill_null not in self.classes_:\n            self.idx_fill_null = len(self.classes_)\n        else:\n            self.idx_fill_null = self.class_to_index_[value_fill_null]\n        return self\n    \n    def transform(self, y):\n        # Gặp những giá trị đó giờ chưa thấy => Xem như fill null\n        return [self.class_to_index_.get(item, self.idx_fill_null) for item in y]\n    \n    def fit_transform(self, y: pd.Series, value_fill_null):\n        self.fit(y, value_fill_null)\n        return self.transform(y)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:33:58.993423Z","iopub.execute_input":"2024-05-17T14:33:58.994306Z","iopub.status.idle":"2024-05-17T14:33:59.004060Z","shell.execute_reply.started":"2024-05-17T14:33:58.994262Z","shell.execute_reply":"2024-05-17T14:33:59.002868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Select feature by Tree-based methods (Random Forest)","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n# model = RandomForestClassifier(random_state=SEED)\n# md = RandomForestClassifier(n_estimators=100, random_state=SEED) # Đổi lấy mô hình có độ chính xác cao nhất bên dưới\ndef getFeatureImportances(model, X, y, features, types):\n    model.fit(X, y)\n    feature_importances = model.feature_importances_\n    # Sắp xếp giảm dẫn\n    feature_importances = [\n        (feature, type, importance) for feature, type, importance in zip(features, types, feature_importances)\n    ]\n    feature_importances.sort(key=lambda x: x[2], reverse=True)\n    return feature_importances","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:34:12.343939Z","iopub.execute_input":"2024-05-17T14:34:12.345129Z","iopub.status.idle":"2024-05-17T14:34:12.352963Z","shell.execute_reply.started":"2024-05-17T14:34:12.345057Z","shell.execute_reply":"2024-05-17T14:34:12.351571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cross validation","metadata":{}},{"cell_type":"code","source":"# final_features = sorted_feature_importance\nfrom sklearn.metrics import make_scorer, roc_auc_score\nfrom sklearn.model_selection import cross_val_score\ndef averageCrossValidation(model, X, y):\n    auc_scorer = make_scorer(roc_auc_score, needs_proba=True)\n    cv_scores = cross_val_score(model, X, y, cv=5, scoring=auc_scorer)\n    return cv_scores.mean()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:34:20.713419Z","iopub.execute_input":"2024-05-17T14:34:20.713740Z","iopub.status.idle":"2024-05-17T14:34:20.720153Z","shell.execute_reply.started":"2024-05-17T14:34:20.713698Z","shell.execute_reply":"2024-05-17T14:34:20.719047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Corr and feature importances","metadata":{}},{"cell_type":"code","source":"MIN_CORR = 0.9\nMAX_CORR = 1\nSTEP_CORR = 0.01\nCORR_THRES_S = np.round(np.arange(MIN_CORR, MAX_CORR + STEP_CORR, STEP_CORR), 2)\n\nMIN_AMOUNT = 155\nMAX_AMOUNT = 175\nSTEP_AMOUNT = 2\nAMOUNT_THRES_S = np.arange(MIN_AMOUNT, MAX_AMOUNT + STEP_AMOUNT, STEP_AMOUNT)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:38:38.076255Z","iopub.execute_input":"2024-05-17T14:38:38.076557Z","iopub.status.idle":"2024-05-17T14:38:38.083763Z","shell.execute_reply.started":"2024-05-17T14:38:38.076527Z","shell.execute_reply":"2024-05-17T14:38:38.082560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_feature2rank = {get_feature(info): rank for rank, info in enumerate(filtered_numerical_summary.rows(named=True))}\ncategorical_feature2rank = {get_feature(info): rank for rank, info in enumerate(filtered_categorical_summary.rows(named=True))}","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:38:45.139197Z","iopub.execute_input":"2024-05-17T14:38:45.139522Z","iopub.status.idle":"2024-05-17T14:38:45.152719Z","shell.execute_reply.started":"2024-05-17T14:38:45.139488Z","shell.execute_reply":"2024-05-17T14:38:45.151555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Encoder\nle = CustomLabelEncoder()\nfor categorical_feature in categorical_features:\n    sr = train_df[categorical_feature]\n    \n    train_df[categorical_feature] = le.fit_transform(sr, feature2null_value[categorical_feature])\n    \n    sr = test_df[categorical_feature]\n    test_df[categorical_feature] = le.transform(sr)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:38:53.474735Z","iopub.execute_input":"2024-05-17T14:38:53.475106Z","iopub.status.idle":"2024-05-17T14:39:50.762943Z","shell.execute_reply.started":"2024-05-17T14:38:53.475043Z","shell.execute_reply":"2024-05-17T14:39:50.760783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df = Pipeline.under_sample(train_df, is_polars=False)\n# print(train_df[\"target\"].value_counts())\n# 47994\n# 47994","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_auc = pd.DataFrame(index=CORR_THRES_S, columns=AMOUNT_THRES_S, dtype=np.float64)\n# Cột: Amount\n# Hàng: corr\nmodel = RandomForestClassifier(random_state=SEED)\ncorr_mtrx = pl.from_pandas(train_df[numerical_features]).corr() # polar nhanh hơn\ny = train_df[\"target\"]\n\nprint(f\"corr_times x amount_times: {len(CORR_THRES_S)} x {len(AMOUNT_THRES_S)} = {len(CORR_THRES_S) * len(AMOUNT_THRES_S)}\")\n\nfor corr_thres in CORR_THRES_S:\n    print(f\"Corr: {corr_thres}\")\n    \n    seleted_numerical_features = getNumericalFeaturesByCorr(\n        corr_mtrx, numerical_features, numerical_feature2rank, corr_thres)\n\n    features = categorical_features + seleted_numerical_features\n    types = [\"categorical\"] * len(categorical_features) + [\"numerical\"] * len(seleted_numerical_features)\n    sorted_feature_importances = getFeatureImportances(model, train_df[features], y, features, types)\n    \n    for amount in AMOUNT_THRES_S:\n        if amount > len(sorted_feature_importances):\n            continue\n            \n        seleted_features = [_[0] for _ in sorted_feature_importances[:amount]]\n        X = train_df[seleted_features].to_numpy()\n\n        auc = averageCrossValidation(model, X, y)\n\n        print(f\"\\tAmount: {amount}: {auc:.4f}\")\n        df_auc.loc[corr_thres, amount] = auc","metadata":{"execution":{"iopub.status.busy":"2024-05-17T14:40:22.923912Z","iopub.execute_input":"2024-05-17T14:40:22.924265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_auc","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df_auc, annot=True)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Post processing\n- Qcut: số lượng q là bao nhiêu ?\n- Xoá outlier\n    - Categorical: với các giá trị có tầng suất xuất hiện thấp -> Đặt là rare -> Tầng xuất bao nhiêu là thấp\n    - Numerical: thực hiện qcut thì khỏi outlier ?","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}