{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# suppress enjoying warnings from seaborn\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\n# for fiding file names\nimport os\nfrom pathlib import Path\nfrom glob import glob\nimport gc\n    \n# data processing libraries\nimport polars as pl\nimport numpy as np\nimport pandas as pd\n\n# for visualizing data\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# sklearn functions\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold\nfrom sklearn.preprocessing import LabelEncoder\n# for validating models\nfrom sklearn.metrics import roc_auc_score, roc_curve\n\n# LightGBM modeling\nimport lightgbm as lgb\n# Có thể thực hiện huấn luyện mô hình với các cách chọn feature khác nhau\n\n# define default colors for plots in notebook\nfrom matplotlib import cycler\nfrom matplotlib.colors import LinearSegmentedColormap\nCOLORS = [\"#068D9D\", \"#53599A\", \"#607BB0\", \"#6D9DC5\", \"#77BECF\", \"#80DED9\", \"#AEECEF\"]\nplt.rc('axes', facecolor='#E6E6E6', edgecolor='none', axisbelow=True, grid=True, prop_cycle=cycler('color', COLORS))\n\n# project CONSTANTS\nROOT = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR = ROOT / \"parquet_files\" / \"test\"\nSEED = 42","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-09T09:19:35.760253Z","iopub.execute_input":"2024-05-09T09:19:35.760600Z","iopub.status.idle":"2024-05-09T09:19:39.607417Z","shell.execute_reply.started":"2024-05-09T09:19:35.760574Z","shell.execute_reply":"2024-05-09T09:19:39.606404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    def set_table_dtypes(df):\n        # Thiếu xử lý feature dạng T và L\n        # Mấy cái đó nó tự áp kiểu dữ liệu vào luôn ?\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n        return df\n    def under_sample(df):\n        df_1 = df.filter(df[\"target\"] == 1)\n        count_per_class = df_1.shape[0]\n        df_0 = df.filter(df[\"target\"] == 0).sample(n=count_per_class, seed=SEED, with_replacement=False)\n        df = pl.concat([df_0, df_1], how=\"vertical_relaxed\")\n        return Pipeline.sort_df(df, 0)\n    def sort_df(df, depth):\n        # Sắp xếp để hàm first có ý nghĩa\n        if depth is None or depth == 0:\n            return df.sort(\"case_id\")\n        elif depth == 1:\n            return df.sort([\"case_id\", \"num_group1\"])\n        else:\n            return df.sort([\"case_id\", \"num_group1\", \"num_group2\"])\n    \n    def handle_dates(df):\n        # Lấy số ngày từ date_decision\n        # Hàm này được gọi khi mà đã thực hiện các phép tổng hợp xong\n        \n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                # Có thể cân nhắc thực hiện thêm các loại đặc trưng khác\n                # Chuyển các cột date thành số ngày so với date decision\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))  #!!?\n                # Hàm này thay đổi giá trị của cột col luôn chứ không thêm cột mới !\n                df = df.with_columns(pl.col(col).dt.total_days()) # t - t-1\n        df = df.drop(\"date_decision\", \"MONTH\")\n        return df\n\n    def filter_cols(df):\n        # Hàm này được áp dụng khi mà đã thực hiện các hàm tổng hợp !\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                # Bỏ các cột có tỉ lệ null trên ???\n                isnull = df[col].is_null().mean()\n                if isnull > 0.3:\n                    df = df.drop(col)\n        \n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                # Bỏ các cột category có số giá trị unique trên 200\n                freq = df[col].n_unique()\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n        \n        return df","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.478263Z","iopub.execute_input":"2024-05-09T02:43:37.479090Z","iopub.status.idle":"2024-05-09T02:43:37.505123Z","shell.execute_reply.started":"2024-05-09T02:43:37.479042Z","shell.execute_reply":"2024-05-09T02:43:37.503852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    def is_num_col(col, df):\n        # Chổ này cần xem xét kiểu dữ liệu để áp dụng phù hợp\n        return ((col[-1] in (\"P\", \"A\") or \n                ((col[-1] in (\"T\", \"L\"))) and df[col].dtype.is_numeric()))\n    def is_str_col(col, df):\n        return ((col[-1] == \"M\") or \n                ((col[-1] in (\"T\", \"L\")) and not df[col].dtype.is_numeric()))\n    def is_date_col(col, df):\n        return col[-1] == \"D\"\n    def agg_first(col):\n        return pl.first(col).alias(f\"first_{col}\"), \"first\"\n    def agg_last(col):\n        return pl.last(col).alias(f\"last_{col}\"), \"last\"\n    def agg_mean(col):\n        return pl.mean(col).alias(f\"mean_{col}\"), \"mean\"\n    def agg_median(col):\n        return pl.median(col).alias(f\"median_{col}\"), \"median\"\n    def agg_max(col):\n        return pl.max(col).alias(f\"max_{col}\"), \"max\"\n    def agg_min(col):\n        return pl.min(col).alias(f\"min_{col}\"), \"min\"\n    def agg_std(col):\n        return pl.std(col).alias(f\"std_{col}\"), \"std\"\n    def agg_mode(col):\n        return pl.col(col).drop_nulls().mode().first().alias(f\"mode_{col}\"), \"mode\"\n    def agg_count(col):\n        return pl.count(col).alias(f\"count_{col}\"), \"count\"\n    def agg_nunique(col):\n        return pl.n_unique(col).alias(f\"nunique_{col}\"), \"nunique\"\n    def agg_sumvalue(col, value):\n        prefix = f\"sum{value}\"\n        return (pl.col(col) == value).sum().alias(f\"{prefix}_{col}\"), prefix\n    def num_expr(df):\n        cols = [col for col in df.columns if Aggregator.is_num_col(col, df)]\n        exprs = []\n        lst_numerical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        lst_categorical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        for col in cols:\n            for expr, agg in [Aggregator.agg_first(col),\n                                  Aggregator.agg_last(col), \n                                  Aggregator.agg_mean(col), \n                                  Aggregator.agg_median(col), \n                                  Aggregator.agg_max(col), \n                                  Aggregator.agg_min(col), \n                                  Aggregator.agg_std(col)]:\n                lst_numerical_feature_agg.append({\"feature\": col, \"agg\": agg})\n                exprs.append(expr)\n        return exprs, {\"numerical\": lst_numerical_feature_agg, \"categorical\": lst_categorical_feature_agg}\n    \n    def date_expr(df):\n        cols = [col for col in df.columns if Aggregator.is_date_col(col, df)]\n        exprs = []\n        lst_numerical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        lst_categorical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        for col in cols:\n            for expr, agg in [Aggregator.agg_first(col),\n                              Aggregator.agg_last(col), \n                              Aggregator.agg_mean(col), \n                              Aggregator.agg_median(col), \n                              Aggregator.agg_max(col), \n                              Aggregator.agg_min(col), \n                              Aggregator.agg_std(col)]:\n                lst_numerical_feature_agg.append({\"feature\": col, \"agg\": agg})\n                exprs.append(expr)\n        return exprs, {\"numerical\": lst_numerical_feature_agg, \"categorical\": lst_categorical_feature_agg}\n    def str_expr(df, name):\n        cols = [col for col in df.columns if Aggregator.is_str_col(col, df)]\n        # Phải đọc được toàn bộ file, toàn bộ chunk để xử lý trường hợp sumvalue\n        paths = filter(lambda _: name in _.name ,TRAIN_DIR.iterdir())\n        \n        exprs = []\n        lst_numerical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        lst_categorical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        for col in cols:\n            # Chổ này nên tách riêng riêng\n            # Nếu có nhiều nhất 30 giá trị duy nhất\n            # Thực hiện các feature: is_value cho tất cả\n            # Nếu có hơn 30 giá trị duy nhất: thực hiện is value cho giá trị xuất hiện nhiều nhất\n            # Đọc riêng ra cột đó\n            sr = pl.Series(dtype=pl.String)\n            for path in paths:\n                sr.extend(pl.read_parquet(path, columns=[col])[col])\n            N_UNIQUE_THRES = 30\n            agg_sumvalues = None\n            sr = sr.drop_nulls()\n            if sr.n_unique() > N_UNIQUE_THRES:\n                mode = sr.mode()[0]\n                expr, agg = Aggregator.agg_sumvalue(col, mode) # Cái này là type: number\n                exprs.append(expr)\n                lst_numerical_feature_agg.append({\"feature\": col, \"agg\": agg})\n            else:\n                for val in sr.unique():\n                    expr, agg = Aggregator.agg_sumvalue(col, val)\n                    exprs.append(expr)\n                    lst_numerical_feature_agg.append({\"feature\": col, \"agg\": agg})\n\n            # nunique\n            expr, agg = Aggregator.agg_nunique(col)\n            exprs.append(expr)\n            lst_numerical_feature_agg.append({\"feature\": col, \"agg\": agg})\n\n            for expr, agg in [Aggregator.agg_first(col),\n                              Aggregator.agg_last(col),\n                              Aggregator.agg_max(col),\n                              Aggregator.agg_min(col),\n                              Aggregator.agg_mode(col)]:\n                exprs.append(expr)\n                lst_categorical_feature_agg.append({\"feature\": col, \"agg\": agg})\n        return exprs, {\"numerical\": lst_numerical_feature_agg, \"categorical\": lst_categorical_feature_agg}\n    \n    def count_expr(df):\n        # Chổ này làm qq gì đây ?\n        cols = [col for col in df.columns if \"num_group\" in col]\n        exprs = []\n        lst_numerical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        lst_categorical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        for col in cols:\n            for expr, agg in [Aggregator.agg_count(col), \n                                  Aggregator.agg_nunique(col), \n                                  Aggregator.agg_max(col), # tương đồng với last: tại có sort\n                                  Aggregator.agg_min(col), # tương đồng với min: tại có sort\n                                  Aggregator.agg_mode(col)]: # mode ở đây hợp lý hơn là mean\n                lst_numerical_feature_agg.append({\"feature\": col, \"agg\": agg})\n                exprs.append(expr)\n        return exprs, {\"numerical\": lst_numerical_feature_agg, \"categorical\": lst_categorical_feature_agg}\n    def summary_1_2(name, depth, lst_feature_agg):\n        features = [feature_agg_type[\"feature\"] for feature_agg_type in lst_feature_agg]\n        aggs = [feature_agg_type[\"agg\"] for feature_agg_type in lst_feature_agg]\n        count = len(lst_feature_agg)\n        return pl.DataFrame({\n            \"name\": [name] * count,\n            \"depth\": [depth] * count,\n            \"feature\": features,\n            \"agg\": aggs},\n            schema=[\n                (\"name\", pl.String),\n                (\"depth\", pl.Int8),\n                (\"feature\", pl.String),\n                (\"agg\", pl.String)])\n    def get_exprs(name, depth):\n        # df: cần được tổng hợp lại\n        path = next(filter(lambda _: name in _.name ,TRAIN_DIR.iterdir()))\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        res_exprs = []\n        res_lst_numerical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        res_lst_categorical_feature_agg = [] # {\"feature\": ,\"agg\": }\n        \n        for exprs, dict_lst_feature_agg in [Aggregator.num_expr(df), \n                                            Aggregator.date_expr(df),\n                                            Aggregator.count_expr(df),\n                                            Aggregator.str_expr(df, name)]:\n            res_exprs += exprs\n            res_lst_numerical_feature_agg += dict_lst_feature_agg[\"numerical\"]\n            res_lst_categorical_feature_agg += dict_lst_feature_agg[\"categorical\"]\n        return res_exprs, {\n            \"numerical\": Aggregator.summary_1_2(name, depth, res_lst_numerical_feature_agg),\n            \"categorical\": Aggregator.summary_1_2(name, depth, res_lst_categorical_feature_agg)}","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.507150Z","iopub.execute_input":"2024-05-09T02:43:37.507941Z","iopub.status.idle":"2024-05-09T02:43:37.555917Z","shell.execute_reply.started":"2024-05-09T02:43:37.507898Z","shell.execute_reply":"2024-05-09T02:43:37.554165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Get_data:\n    def get_base_data():\n        df = pl.read_parquet(TRAIN_DIR / \"train_base.parquet\")\n        df = df.pipe(Pipeline.set_table_dtypes).pipe(Pipeline.under_sample)\n        cols = {\n            \"month_decision\": pl.col(\"date_decision\").dt.month().alias(\"month_decision\"),\n            \"weekday_decision\": pl.col(\"date_decision\").dt.weekday().alias(\"weekday_decision\"),\n            \"year_decision\": pl.col(\"date_decision\").dt.year().alias(\"year_decision\")\n        }\n        df = df.with_columns(cols.values())\n        count = len(cols)\n        schema = [(\"name\", pl.String),(\"depth\", pl.Int8),(\"feature\", pl.String),(\"agg\", pl.String)]\n        numerical_summary = pl.DataFrame({\n            \"name\": [\"base\"] * count,\n            \"depth\": [0] * count,\n            \"feature\": cols.keys(),\n            \"agg\": [None] * count,\n            },schema=schema)\n        categorical_summary = pl.DataFrame(schema=schema)\n        return df, {\"numerical\": numerical_summary, \"categorical\": categorical_summary}\n    def summary_0(name, df):\n        features = df.columns\n        features.remove(\"case_id\")\n        \n        numerical_features = list(filter(lambda feature: Aggregator.is_num_col(feature, df) or Aggregator.is_date_col(feature, df), features))\n        categorical_features = list(filter(lambda feature: Aggregator.is_str_col(feature, df), features))\n        numerical_features_count = len(numerical_features)\n        categorical_features_count = len(categorical_features)\n\n        schema = [(\"name\", pl.String), (\"depth\", pl.Int8), (\"feature\", pl.String), (\"agg\", pl.String)]\n        numerical_summary = pl.DataFrame({\n            \"name\": [name] * numerical_features_count,\n            \"depth\": [depth] * numerical_features_count,\n            \"feature\": numerical_features,\n            \"agg\": [None] * numerical_features_count},\n            schema=schema)\n        categorical_summary = pl.DataFrame({\n            \"name\": [name] * categorical_features_count,\n            \"depth\": [depth] * categorical_features_count,\n            \"feature\": categorical_features,\n            \"agg\": [None] * categorical_features_count},\n            schema=schema)\n        \n        return {\"numerical\": numerical_summary, \"categorical\": categorical_summary}\n    def get_data(df_base, name, depth=None):\n        \n        paths = filter(lambda _: name in _.name ,TRAIN_DIR.iterdir())\n        \n        if depth == 1 or depth == 2:\n            exprs, summary = Aggregator.get_exprs(name, depth) # summary: {\"numerical\": numerical_summary, \"categorical\": categorical_summary}\n        chunks = []\n        for path in paths:\n            df = pl.read_parquet(path)\n            df = df.pipe(Pipeline.set_table_dtypes)\n            df = Pipeline.sort_df(df, depth)\n            if depth == 1 or depth == 2:\n                df = df.group_by(\"case_id\").agg(exprs)\n            chunks.append(df)\n        df = pl.concat(chunks, how=\"vertical_relaxed\")\n        # relaxed: chuyển đổi để cùng kiểu dữ liệu giữa các hàng\n        if depth == 0:\n            summary = Get_data.summary_0(name, df)\n\n        # Thực hiện join table\n        df = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_???\") # Demo: kiểm tra cái suffix này\n\n        # Thực hiện feature eng: đổi các kiểu dữ liệu date\n        df = df.pipe(Pipeline.handle_dates)\n        return df, summary","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.560868Z","iopub.execute_input":"2024-05-09T02:43:37.562255Z","iopub.status.idle":"2024-05-09T02:43:37.592965Z","shell.execute_reply.started":"2024-05-09T02:43:37.562201Z","shell.execute_reply":"2024-05-09T02:43:37.591659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Fill_null:\n    def get_zero():\n        return 0, \"zero\"\n    def get_median(sr:pl.Series):\n        return sr.median(), \"median\"\n    def get_mean(sr:pl.Series):\n        return sr.mean(), \"mean\"\n    def get_max(sr:pl.Series):\n        return sr.max(), \"max\"\n    def get_min(sr:pl.Series):\n        return sr.min(), \"min\"\n    def get_mode(sr:pl.Series):\n        return sr.mode()[0], \"mode\"\n    def get_null():\n        return \"null\", \"null\"\n    def make_summary(info, methods):\n        method_count = len(methods)\n        return pl.DataFrame({\n                \"name\": [info[\"name\"]] * method_count,\n                \"depth\": [info[\"depth\"]] * method_count,\n                \"feature\": [info[\"feature\"]] * method_count,\n                \"agg\": [info[\"agg\"]] * method_count,\n                \"null_percent\": [info[\"null_percent\"]] * method_count,\n                \"fill_null_method\": methods\n            },\n            schema=[\n                (\"name\", pl.String),\n                (\"depth\", pl.Int8),\n                (\"feature\", pl.String),\n                (\"agg\", pl.String),\n                (\"null_percent\", pl.Float32),\n                (\"fill_null_method\", pl.String)])\n    def get_numerical_values(info, sr:pl.Series): #\"name\", \"depth\", \"feature\", \"agg\", \"null_percent\"\n        values, methods = [], []\n        sr = sr.drop_nulls()\n        for value, method in [Fill_null.get_zero(),\n                              Fill_null.get_median(sr),\n                              Fill_null.get_mean(sr),\n                              Fill_null.get_max(sr),\n                              Fill_null.get_min(sr)]:\n            if value is None:\n                continue\n            values.append(value)\n            methods.append(method)\n        summary = Fill_null.make_summary(info, methods)\n        return values, summary\n    def get_categorical_values(info, sr:pl.Series): #\"name\", \"depth\", \"feature\", \"agg\", \"null_percent\"\n        values, methods = [], []\n        sr = sr.drop_nulls()\n        for value, method in [Fill_null.get_null(), Fill_null.get_mode(sr)]:\n            if value is None:\n                continue\n            values.append(value)\n            methods.append(method)\n        summary = Fill_null.make_summary(info, methods)\n        return values, summary","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.595109Z","iopub.execute_input":"2024-05-09T02:43:37.595669Z","iopub.status.idle":"2024-05-09T02:43:37.618489Z","shell.execute_reply.started":"2024-05-09T02:43:37.595623Z","shell.execute_reply":"2024-05-09T02:43:37.617199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import mutual_info_classif\nfrom sklearn.preprocessing import LabelEncoder\nimport sklearn\nclass Scoring:\n    def get_information_gain(sr, target):\n        return mutual_info_classif(\n            np.expand_dims(sr.to_numpy(), axis=1), \n            target,\n            random_state=SEED)[0]\n    def get_nunique(sr):\n        return sr.n_unique()\n    def get_std(sr):\n        return sr.std()\n    def get_chi2_p_value(sr, target):\n        le = LabelEncoder()\n        arr = le.fit_transform(sr)\n        chi2, p_values = sklearn.feature_selection.chi2(\n            np.expand_dims(arr, axis=1),\n            target)\n        return {\n            \"chi2\": chi2[0],\n            \"p_value\": p_values[0]\n        }\n    def get_numerical_score(sr, target):\n        return {\n            \"information_gain\": Scoring.get_information_gain(sr, target),\n            \"std\": Scoring.get_std(sr),\n        }\n    def get_categorical_score(sr, target):\n        scores = Scoring.get_chi2_p_value(sr, target)\n        scores[\"nunique\"] = Scoring.get_nunique(sr)\n        return scores","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.620048Z","iopub.execute_input":"2024-05-09T02:43:37.620833Z","iopub.status.idle":"2024-05-09T02:43:37.881038Z","shell.execute_reply.started":"2024-05-09T02:43:37.620767Z","shell.execute_reply":"2024-05-09T02:43:37.879385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_null_percent(sr: pl.Series):\n    return sr.is_null().mean() * 100\ndef Fill_null_and_Scoring_sr(sr, target, row_info, type):\n    row_info = row_info.copy()\n    if sr.drop_nulls().n_unique() <= 1: # tất cả là null, hoặc chỉ có 1 giá trị\n        return None\n    elif sr.is_null().not_().all(): # nếu không có null\n        # sửa\n        row_info[\"fill_null_method\"] = None\n        \n        if type == \"numerical\":\n            row_info.update(Scoring.get_numerical_score(sr, target))\n            summary = pl.DataFrame(row_info, schema=numerical_schema)\n        else:\n            row_info.update(Scoring.get_categorical_score(sr, target))\n            summary = pl.DataFrame(row_info, schema=categorical_schema)\n    else:\n        if type == \"numerical\":\n            scores = {\"std\": [], \"information_gain\": []}\n            values, _summary = Fill_null.get_numerical_values(row_info, sr) # no: \"score\"\n        else:\n            scores = {\"nunique\": [], \"chi2\": [], \"p_value\": []}\n            values, _summary = Fill_null.get_categorical_values(row_info, sr) # no: \"score\"\n        for value in values:\n            # Dùng series với giá trị trả về là value cần fill cũng được\n            sr_filled_null = sr.fill_null(value)\n            # Scoring\n            # Xử lý\n            if type == \"numerical\":\n                _scores = Scoring.get_numerical_score(sr_filled_null, target)\n            else:\n                _scores = Scoring.get_categorical_score(sr_filled_null, target)\n            \n            for metric in _scores:\n                scores[metric].append(_scores[metric])\n            \n        score_cols = [pl.Series(scores[metric], dtype=pl.Float32).alias(metric) for metric in scores]\n        summary = _summary.with_columns(score_cols)\n    return summary\ndef Fill_null_and_Scoring_df(df, __summary):\n    res_numerical_summary = pl.DataFrame(schema=numerical_schema)\n    res_categorical_summary = pl.DataFrame(schema=categorical_schema)\n    # fill null\n    target = df[\"target\"]\n    for type in [\"numerical\", \"categorical\"]:\n        for row_info in __summary[type].rows(named=True):\n            sr = None\n            depth = row_info[\"depth\"]\n            if depth == 0:\n                sr = df[row_info[\"feature\"]]\n            elif depth == 1 or depth == 2:\n                agg_feature = f\"{row_info['agg']}_{row_info['feature']}\"\n                sr = df[agg_feature]\n            \n            \n            row_info[\"null_percent\"] = get_null_percent(sr)\n\n            summary = Fill_null_and_Scoring_sr(sr, target, row_info, type)\n            if summary is not None: # None if: feature contain all null\n                if type == \"numerical\":\n                    res_numerical_summary.extend(summary)\n                else:\n                    res_categorical_summary.extend(summary)\n    return {\"numerical\": res_numerical_summary, \"categorical\": res_categorical_summary}","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.883183Z","iopub.execute_input":"2024-05-09T02:43:37.884192Z","iopub.status.idle":"2024-05-09T02:43:37.904734Z","shell.execute_reply.started":"2024-05-09T02:43:37.884139Z","shell.execute_reply":"2024-05-09T02:43:37.903609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Này là những gì có thôi\nnames = {\n    0: [\"base\", \"static_cb_0\", \"static_0\"],\n    1: [\"applprev_1\",\n        \"tax_registry_a_1\",\n        \"tax_registry_b_1\",\n        \"tax_registry_c_1\",\n        \"credit_bureau_a_1\",\n        \"credit_bureau_b_1\",\n        \"other_1\", \n        \"person_1\", \n        \"deposit_1\", \n        \"debitcard_1\"],\n    2: [\"applprev_2\",\n        \"credit_bureau_b_2\",\n        \"credit_bureau_a_2\",\n        \"person_2\"]\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.906849Z","iopub.execute_input":"2024-05-09T02:43:37.907811Z","iopub.status.idle":"2024-05-09T02:43:37.922765Z","shell.execute_reply.started":"2024-05-09T02:43:37.907759Z","shell.execute_reply":"2024-05-09T02:43:37.921793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_schema = (\n    (\"name\", pl.String),\n    (\"depth\", pl.Int8),\n    (\"feature\", pl.String),\n    (\"agg\", pl.String),\n    (\"null_percent\", pl.Float32),\n    (\"fill_null_method\", pl.String),\n    (\"std\", pl.Float32),\n    (\"information_gain\", pl.Float32))\ncategorical_schema = (\n    (\"name\", pl.String),\n    (\"depth\", pl.Int8),\n    (\"feature\", pl.String),\n    (\"agg\", pl.String),\n    (\"null_percent\", pl.Float32),\n    (\"fill_null_method\", pl.String),\n    (\"nunique\", pl.Float32),\n    (\"chi2\", pl.Float32),\n    (\"p_value\", pl.Float32))","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.925352Z","iopub.execute_input":"2024-05-09T02:43:37.926334Z","iopub.status.idle":"2024-05-09T02:43:37.939926Z","shell.execute_reply.started":"2024-05-09T02:43:37.926267Z","shell.execute_reply":"2024-05-09T02:43:37.938634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_numerical_summary = pl.DataFrame(schema=numerical_schema)\nres_categorical_summary = pl.DataFrame(schema=categorical_schema)\nprint(res_numerical_summary)\nprint(res_categorical_summary)\n\n# Đọc cái file base trước\n# Tạm thời là không chơi tới chổ fill na method\nname = \"base\"\nprint(name)\ndf_base, __summary = Get_data.get_base_data()\nsummary = Fill_null_and_Scoring_df(df_base, __summary)\nres_numerical_summary.extend(summary[\"numerical\"])\nres_categorical_summary.extend(summary[\"categorical\"])\n\nfor depth, file_names in names.items():\n    print(depth)\n    for name in file_names:\n        if name == \"base\":\n            continue\n        print(\"\\t\", name)\n        df, __summary = Get_data.get_data(df_base, name, depth) # {\"numerical\": numerical_summary, \"categorical\": categorical_summary}, \"name\", \"depth\", \"feature\", \"agg\"\n        \n        summary = Fill_null_and_Scoring_df(df, __summary) # {\"numerical\": numerical_summary, \"categorical\": categorical_summary}\n        res_numerical_summary.extend(summary[\"numerical\"])\n        res_categorical_summary.extend(summary[\"categorical\"])","metadata":{"execution":{"iopub.status.busy":"2024-05-09T02:43:37.944343Z","iopub.execute_input":"2024-05-09T02:43:37.944861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\ndef sub_sample(scores):\n    step = len(scores) / 100\n    _scores = []\n    idx = 0.0\n    while idx < len(scores):\n        _scores.append(scores[int(idx)])\n        idx += step\n    return _scores","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_numerical_summary.to_pandas().to_excel('numerical_summary.xlsx', index=False)\nres_numerical_summary.write_parquet('numerical_summary.parquet')\n\nres_categorical_summary.to_pandas().to_excel('categorical_summary.xlsx', index=False)\nres_categorical_summary.write_parquet('categorical_summary.parquet')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number\n# Phân phối điểm\n# (\"std\", pl.Float32)\n# (\"information_gain\", pl.Float32)\n\nfig, axes = plt.subplots(2, 2, figsize=(2 * 5, 2 * 5))\n\nscores = sub_sample(res_numerical_summary[\"information_gain\"].sort().to_list())\nax = axes[0][0]\nax.plot(scores)\nax.set_xticks([])\nax.set_title(\"information gain\")\n\nscores = sub_sample(res_numerical_summary[\"std\"].sort().to_list())\nax = axes[0][1]\nax.plot(scores)\nax.set_xticks([])\nax.set_title(\"std\")\n\nax = axes[1][0]\nsns.scatterplot(res_numerical_summary, x=\"information_gain\", y=\"std\",size=1, ax=ax)\nax.set_title(\"information_gain x std\")\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Category\n# Phân phối điểm\n# (\"nunique\", pl.Float32)\n# (\"chi2\", pl.Float32),\n# (\"p_value\", pl.Float32)\n\nfig, axes = plt.subplots(2, 3, figsize=(3 * 5, 2 * 5))\n\nfor i, metric in enumerate([\"nunique\", \"chi2\", \"p_value\"]):\n    scores = sub_sample(res_categorical_summary[metric].sort().to_list())\n    ax = axes[0][i]\n    ax.plot(scores)\n    ax.set_xticks([])\n    ax.set_title(metric)\nfor i, (metric1, metric2) in enumerate([(\"nunique\", \"chi2\"), (\"nunique\", \"p_value\"), (\"chi2\", \"p_value\")]):\n    ax = axes[1][i]\n    sns.scatterplot(res_categorical_summary, x=metric1, y=metric2,size=1, ax=ax)\n    ax.set_title(f\"{metric1} x {metric2}\")\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}