{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nFirst and foremost, it is important to note that the following attempt will succeed for training data, but **fail in submission**. Please be aware that this notebook is to warn you not to repeat such a failure.\n\nIn this competition, WEEK_NUM is very important variable and hidden in test dataset.\n\nBelow, we are going to try to predict WEEK_NUM based on the available information in test dataset.\n\nThe strategy in this notebook is:\n1. Find the features highly correlated with WEEK_NUM\n2. Build LinearRegression models using these features (custom pipeline to ignore missing values)\n3. Predict WEEK_NUM using the results of LinearRegression models","metadata":{}},{"cell_type":"code","source":"import joblib\nimport gc\nimport lightgbm as lgb\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport warnings\n\nfrom catboost import CatBoostClassifier, Pool\nfrom glob import glob\nfrom IPython.display import display\nfrom pathlib import Path\nfrom sklearn.base import BaseEstimator, ClassifierMixin\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom typing import Any\ngc.collect()\nwarnings.filterwarnings('ignore')\n\nROOT      = Path('/kaggle/input/home-credit-credit-risk-model-stability')\nTRAIN_DIR = ROOT / 'parquet_files' / 'train'\nTEST_DIR  = ROOT / 'parquet_files' / 'test'","metadata":{"execution":{"iopub.status.busy":"2024-05-19T13:48:58.450002Z","iopub.execute_input":"2024-05-19T13:48:58.450781Z","iopub.status.idle":"2024-05-19T13:49:04.088538Z","shell.execute_reply.started":"2024-05-19T13:48:58.450740Z","shell.execute_reply":"2024-05-19T13:49:04.087325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define data pipeline utilities","metadata":{}},{"cell_type":"code","source":"class Utility:\n    @staticmethod\n    def get_feat_defs(ending_with:str):\n        feat_defs:pl.DataFrame = pl.read_csv(ROOT / 'feature_definitions.csv')\n\n        filtered_feats:pl.DataFrame = feat_defs.filter(pl.col('Variable').apply(lambda var: var.endswith(ending_with)))\n\n        with pl.Config(fmt_str_lengths=200, tbl_rows=-1):\n            print(filtered_feats)\n\n        filtered_feats = None\n        feat_defs = None\n\n     \n    @staticmethod\n    def find_index(lst:list, item:Any) -> int | None:\n        try:\n            return lst.index(item)\n        except ValueError:\n            return None\n\n    \n    @staticmethod\n    def dtype_to_str(dtype:pl.DataType) -> str:\n        dtype_map = {\n            pl.Decimal: 'Decimal',\n\n            pl.Float32: 'Float32',\n            pl.Float64: 'Float64',\n\n            pl.UInt8: 'UInt8',\n            pl.UInt16: 'UInt16',\n            pl.UInt32: 'UInt32',\n            pl.UInt64: 'UInt64',\n\n            pl.Int8: 'Int8',\n            pl.Int16: 'Int16',\n            pl.Int32: 'Int32',\n            pl.Int64: 'Int64',\n\n            pl.Date: 'Date',\n            pl.Datetime: 'Datetime',\n            pl.Duration: 'Duration',\n            pl.Time: 'Time',\n\n            pl.Array: 'Array',\n            pl.List: 'List',\n            pl.Struct: 'Struct',\n\n            pl.String: 'String',\n            pl.Categorical: 'Categorical',\n            pl.Enum: 'Enum',\n            pl.Utf8: 'Utf8',\n\n            pl.Binary: 'Binary',\n            pl.Boolean: 'Boolean',\n            pl.Null: 'Null',\n            pl.Object: 'Object',\n            pl.Unknown: 'Unknown'\n        }\n\n        return dtype_map.get(dtype)\n\n    \n    @staticmethod\n    def find_feat_occur(regex_path:str, ending_with:str) -> pl.DataFrame:\n        feat_defs:pl.DataFrame = pl.read_csv(ROOT / 'feature_definitions.csv').filter(pl.col('Variable').apply(lambda var: var.endswith(ending_with)))\n        feat_defs.sort(by=['Variable'])\n\n        feats:list = feat_defs['Variable'].to_list()\n        feats.sort()\n\n        occurrences:list = [[set(), set()] for _ in range(feat_defs.height)]\n\n        for path in glob(str(regex_path)):\n            df_schema:dict = pl.read_parquet_schema(path)\n\n            for (feat, dtype) in df_schema.items():\n                index:int = Utility.find_index(feats, feat)\n                if index != None:\n                    occurrences[index][0].add(Utility.dtype_to_str(dtype))\n                    occurrences[index][1].add(Path(path).stem)\n\n        data_types:list[str] = [None] * feat_defs.height\n        file_locs:list[str] = [None] * feat_defs.height\n\n        for i, feat in enumerate(feats):\n            data_types[i] = list(occurrences[i][0])\n            file_locs[i] = list(occurrences[i][1])\n\n        feat_defs = feat_defs.with_columns(pl.Series(data_types).alias('Data_Type(s)'))\n        feat_defs = feat_defs.with_columns(pl.Series(file_locs).alias('File_Loc(s)'))\n\n        return feat_defs\n    \n    \n    def reduce_memory_usage(df:pl.DataFrame, name) -> pl.DataFrame:\n        print(f'Memory usage of dataframe \\'{name}\\' is {round(df.estimated_size(\"mb\"), 2)} MB.')\n\n        int_types = [pl.Int8, pl.Int16, pl.Int32, pl.Int64, pl.UInt8, pl.UInt16, pl.UInt32, pl.UInt64]\n        float_types = [pl.Float32, pl.Float64]\n\n        for col in df.columns:\n            col_type = df[col].dtype\n            if (col_type in int_types + float_types):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if c_min is not None and c_max is not None:\n                    if col_type in int_types:\n                        if c_min >= 0:\n                            if c_min >= np.iinfo(np.uint8).min and c_max <= np.iinfo(np.uint8).max:\n                                df = df.with_columns(df[col].cast(pl.UInt8))\n                            elif c_min >= np.iinfo(np.uint16).min and c_max <= np.iinfo(np.uint16).max:\n                                df = df.with_columns(df[col].cast(pl.UInt16))\n                            elif c_min >= np.iinfo(np.uint32).min and c_max <= np.iinfo(np.uint32).max:\n                                df = df.with_columns(df[col].cast(pl.UInt32))\n                            elif c_min >= np.iinfo(np.uint64).min and c_max <= np.iinfo(np.uint64).max:\n                                df = df.with_columns(df[col].cast(pl.UInt64))\n                        else:\n                            if c_min >= np.iinfo(np.int8).min and c_max <= np.iinfo(np.int8).max:\n                                df = df.with_columns(df[col].cast(pl.Int8))\n                            elif c_min >= np.iinfo(np.int16).min and c_max <= np.iinfo(np.int16).max:\n                                df = df.with_columns(df[col].cast(pl.Int16))\n                            elif c_min >= np.iinfo(np.int32).min and c_max <= np.iinfo(np.int32).max:\n                                df = df.with_columns(df[col].cast(pl.Int32))\n                            elif c_min >= np.iinfo(np.int64).min and c_max <= np.iinfo(np.int64).max:\n                                df = df.with_columns(df[col].cast(pl.Int64))\n                    elif col_type in float_types:\n                        if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                            df = df.with_columns(df[col].cast(pl.Float32))\n\n        print(f'Memory usage of dataframe \\'{name}\\' became {round(df.estimated_size(\"mb\"), 4)} MB.')\n\n        return df\n\n\n    def to_pandas(df:pl.DataFrame, cat_cols:list[str]=None) -> (pd.DataFrame, list[str]):\n        df:pd.DataFrame = df.to_pandas()\n\n        if cat_cols is None:\n            cat_cols = list(df.select_dtypes('object').columns)\n\n        df[cat_cols] = df[cat_cols].astype('str')\n\n        return df, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-05-19T13:49:04.090666Z","iopub.execute_input":"2024-05-19T13:49:04.091255Z","iopub.status.idle":"2024-05-19T13:49:04.130374Z","shell.execute_reply.started":"2024-05-19T13:49:04.091223Z","shell.execute_reply":"2024-05-19T13:49:04.128893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def max_expr(df:pl.LazyFrame) -> list[pl.Series]:\n        cols:list[str] = [col for col in df.columns if (col[-1] in ('P', 'M', 'A', 'D', 'T', 'L')) or ('num_group' in col)]\n\n        expr_max:list[pl.Series] = [pl.col(col).max().alias(f'max_{col}') for col in cols]\n\n        return expr_max\n    \n    \n    @staticmethod\n    def min_expr(df:pl.LazyFrame) -> list[pl.Series]:\n        cols:list[str] = [col for col in df.columns if (col[-1] in ('P', 'M', 'A', 'D', 'T', 'L')) or ('num_group' in col)]\n\n        expr_min:list[pl.Series] = [pl.col(col).min().alias(f'min_{col}') for col in cols]\n\n        return expr_min\n    \n    \n    @staticmethod\n    def mean_expr(df:pl.LazyFrame) -> list[pl.Series]:\n        cols:list[str] = [col for col in df.columns if col.endswith(('P', 'A', 'D'))]\n\n        expr_mean:list[pl.Series] = [pl.col(col).mean().alias(f'mean_{col}') for col in cols]\n\n        return expr_mean\n    \n    \n    @staticmethod\n    def var_expr(df:pl.LazyFrame) -> list[pl.Series]:\n        cols:list[str] = [col for col in df.columns if col.endswith(('P', 'A', 'D'))]\n\n        expr_mean:list[pl.Series] = [pl.col(col).var().alias(f'var_{col}') for col in cols]\n\n        return expr_mean\n    \n    \n    @staticmethod\n    def mode_expr(df:pl.LazyFrame) -> list[pl.Series]:\n        cols:list[str] = [col for col in df.columns if col.endswith('M')]\n\n        expr_mode:list[pl.Series] = [pl.col(col).drop_nulls().mode().first().alias(f'mode_{col}') for col in cols]\n\n        return expr_mode\n\n    @staticmethod\n    def get_exprs(df:pl.LazyFrame) -> list[pl.Series]:\n        exprs = Aggregator.max_expr(df) + \\\n                Aggregator.mean_expr(df) + \\\n                Aggregator.var_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-05-19T13:49:04.131564Z","iopub.execute_input":"2024-05-19T13:49:04.132466Z","iopub.status.idle":"2024-05-19T13:49:04.161690Z","shell.execute_reply.started":"2024-05-19T13:49:04.132435Z","shell.execute_reply":"2024-05-19T13:49:04.160272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SchemaGen:\n    @staticmethod\n    def change_dtypes(df:pl.LazyFrame) -> pl.LazyFrame:\n        for col in df.columns:\n            if col == 'case_id':\n                df = df.with_columns(pl.col(col).cast(pl.UInt32).alias(col))\n            elif col in ['WEEK_NUM', 'num_group1', 'num_group2']:\n                df = df.with_columns(pl.col(col).cast(pl.UInt16).alias(col))\n            elif col == 'date_decision' or col[-1] == 'D':\n                df = df.with_columns(pl.col(col).cast(pl.Date).alias(col))\n            elif col[-1] in ['P', 'A']:\n                df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n            elif col[-1] in ('M',):\n                    df = df.with_columns(pl.col(col).cast(pl.String));\n        return df\n\n\n    @staticmethod\n    def scan_files(glob_path: str, depth: int = None) -> pl.LazyFrame:\n        chunks: list[pl.LazyFrame] = []\n        for path in glob(str(glob_path)):\n            df: pl.LazyFrame = pl.scan_parquet(path, low_memory=True, rechunk=True).pipe(SchemaGen.change_dtypes)\n            print(f'File {Path(path).stem} loaded into memory.')\n            \n            if depth in (1, 2):\n                exprs: list[pl.Series] = Aggregator.get_exprs(df)\n                df = df.group_by('case_id').agg(exprs)\n\n                del exprs\n                gc.collect()\n                \n            chunks.append(df)\n\n        df: pl.LazyFrame = pl.concat(chunks, how='vertical_relaxed')\n        \n        del chunks\n        gc.collect()\n                \n        df = df.unique(subset=['case_id'])\n    \n        return df\n    \n    \n    @staticmethod\n    def join_dataframes(df_base: pl.LazyFrame, depth_0: list[pl.LazyFrame], depth_1: list[pl.LazyFrame], depth_2: list[pl.LazyFrame]) -> pl.DataFrame:\n        for (i, df) in enumerate(depth_0 + depth_1 + depth_2):\n            df_base = df_base.join(df, how='left', on='case_id', suffix=f'_{i}')\n\n        return df_base.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-19T13:49:04.164897Z","iopub.execute_input":"2024-05-19T13:49:04.165391Z","iopub.status.idle":"2024-05-19T13:49:04.185906Z","shell.execute_reply.started":"2024-05-19T13:49:04.165348Z","shell.execute_reply":"2024-05-19T13:49:04.184682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def handle_dates(df:pl.DataFrame) -> pl.DataFrame:\n    for col in df.columns:\n        if col.endswith('D'):\n            df = df.with_columns(pl.col(col) - pl.col('date_decision'))\n            df = df.with_columns(pl.col(col).dt.total_days().cast(pl.Int32))\n\n    df = df.with_columns([pl.col('date_decision').dt.year().alias('year').cast(pl.Int16), pl.col('date_decision').dt.month().alias('month').cast(pl.UInt8), pl.col('date_decision').dt.weekday().alias('week_num').cast(pl.UInt8)])\n\n    return df.drop('date_decision', 'MONTH', 'WEEK_NUM');\n\ndef choose_dates(df:pl.DataFrame) -> pl.DataFrame:\n    return df.select([\"case_id\", \"WEEK_NUM\"] + [col for col in df.columns if col.endswith('D')]);\n\ndef filter_cols(df:pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:\n        if col not in ['case_id', 'year', 'month', 'week_num', 'target']:\n            null_pct = df[col].is_null().mean()\n\n            if null_pct > 0.95:\n                df = df.drop(col)\n\n    for col in df.columns:\n        if (col not in ['case_id', 'year', 'month', 'week_num', 'target']) & (df[col].dtype == pl.String):\n            freq = df[col].n_unique()\n\n            if (freq > 200) | (freq == 1):\n                df = df.drop(col)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-05-19T13:49:04.187097Z","iopub.execute_input":"2024-05-19T13:49:04.187778Z","iopub.status.idle":"2024-05-19T13:49:04.205016Z","shell.execute_reply.started":"2024-05-19T13:49:04.187749Z","shell.execute_reply":"2024-05-19T13:49:04.203921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndata_store:dict = {\n    'df_base': SchemaGen.scan_files(TRAIN_DIR / 'train_base.parquet'),\n    'depth_0': [\n        SchemaGen.scan_files(TRAIN_DIR / 'train_static_cb_0.parquet'),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_static_0_*.parquet'),\n    ],\n    'depth_1': [\n        SchemaGen.scan_files(TRAIN_DIR / 'train_applprev_1_*.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_tax_registry_a_1.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_tax_registry_b_1.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_tax_registry_c_1.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_credit_bureau_a_1_*.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_credit_bureau_b_1.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_other_1.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_person_1.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_deposit_1.parquet', 1),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_debitcard_1.parquet', 1),\n    ],\n    'depth_2': [\n        SchemaGen.scan_files(TRAIN_DIR / 'train_credit_bureau_a_2_*.parquet', 2),\n        SchemaGen.scan_files(TRAIN_DIR / 'train_credit_bureau_b_2.parquet', 2),\n    ]\n}\n\nfrom datetime import datetime\nstart_date = datetime(2020, 1, 1) # can be anything\ndef convert_date_to_week(df:pl.DataFrame) -> pl.DataFrame:\n    for col in df.columns:\n        if col.endswith('D'):\n#             df = df.with_columns((pl.col(col) - pl.lit(start_date)).dt.days() / 7)\n            df = df.with_columns((pl.col(col) - pl.lit(start_date)).dt.days())\n    return df\n    \n# df_date:pl.LazyFrame = SchemaGen.join_dataframes(**data_store).pipe(choose_dates).pipe(filter_cols).sort(\"case_id\")\ndf_date:pl.LazyFrame = SchemaGen.join_dataframes(**data_store).pipe(choose_dates).pipe(filter_cols).sort(\"case_id\").pipe(convert_date_to_week).pipe(Utility.reduce_memory_usage, \"df_date\")\n\ndel data_store\ngc.collect()\n\nprint(f'Date data shape: {df_date.shape}')\ndisplay(df_date.head(10))","metadata":{"execution":{"iopub.status.busy":"2024-05-19T13:49:04.206320Z","iopub.execute_input":"2024-05-19T13:49:04.207188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_date.head().to_pandas()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA for WEEK_NUM","metadata":{}},{"cell_type":"code","source":"def calculate_correlations(df, column):\n#     print(set([df[col].dtype for col in df.columns]))\n    correlations = {col: df.select(pl.corr(column, col)).to_pandas().iat[0, 0] for col in df.columns if df[col].dtype in (pl.Int16, pl.Int32)}\n#     print(correlations)\n    sorted_correlations = sorted(correlations.items(), key=lambda x: x[1], reverse=True)\n\n    return sorted_correlations\n\n\ncorrelations = calculate_correlations(df_date, 'WEEK_NUM')\n\nfor i, x in enumerate(correlations):\n    col, corr = x\n    print(f\"Column: {col}, Correlation: {corr}\")\n    if i > 20:\n        break\n\nfor col, corr in correlations:\n    if corr > 0.9:\n        print(col)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_corr = [\n    \"max_refreshdate_3813885D\",\n    \"responsedate_1012D\",\n    \"max_recorddate_4527225D\",\n    \"mean_recorddate_4527225D\",\n    \"responsedate_4527233D\",\n    \"assignmentdate_4527235D\",\n    \"max_lastupdate_1112D\",\n    \"mean_lastupdate_1112D\",\n    \"responsedate_4917613D\",\n    \"mean_refreshdate_3813885D\",\n    \"validfrom_1069D\",\n    \"max_deductiondate_4917603D\",\n    \"max_processingdate_168D\",\n    \"mean_processingdate_168D\",\n    \"mean_deductiondate_4917603D\",    \n]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We found the columns highly correlated with WEEK_NUM.\n\nHere, let me assume it's also true in test dataset. So, let's try to build LinearRegression model for each highly correlated column and add the predictions as new features below.","metadata":{}},{"cell_type":"markdown","source":"# Build LinearRegression models","metadata":{}},{"cell_type":"markdown","source":"I found many missing values, so it's insufficient to predict with only one feature. Here, I build a model based on multiple LinearReggression models which utilizes as many non-missing data as possible. ","metadata":{}},{"cell_type":"code","source":"df_date = df_date.to_pandas()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.linear_model import LinearRegression\n\nclass LRPredEncoder(BaseEstimator, TransformerMixin):\n    \"\"\"\n    LinearRegression for each column and use the predictions as a new column.\n    This custom model ignores missing values.\n    \"\"\"\n    def __init__(self):\n        self.lr_models = {}\n\n    def fit(self, X, y):\n        for column in X.columns:\n            non_nan_rows = X[column].notna()\n            X_non_nan = X.loc[non_nan_rows, [column]]\n            y_non_nan = y[non_nan_rows]\n            model = LinearRegression().fit(X_non_nan, y_non_nan)\n            self.lr_models[column] = model\n        return self\n\n    def transform(self, X):\n        X_pred = X.copy()\n        for column, model in self.lr_models.items():\n            non_nan_rows = X[column].notna()\n            X_non_nan = X.loc[non_nan_rows, [column]]\n            X_pred.loc[non_nan_rows, \"pred_\" + column] = model.predict(X_non_nan)\n        return X_pred\n\n    def get_feature_names(self):\n        return [\"pred_\" + column for column in self.lr_models.keys()]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_date.drop(columns=['WEEK_NUM', 'case_id'])\ny = df_date['WEEK_NUM']\nX = X[cols_corr]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc = LRPredEncoder()\nenc.fit(X, y)\nX_transform = enc.transform(X)\nX_transform.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc.get_feature_names()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfor c in enc.get_feature_names():\n    non_nan_rows = X_transform[c].notna()\n    rmse_score = mean_squared_error(y[non_nan_rows], X_transform.loc[non_nan_rows, c], squared=False)\n    print(c, \"\\t\", rmse_score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We fitted LinearRegression models and obtained new features, each of which already predicts WEEK_NUM accurately.\n\nLet's create WEEK_NUM_pred column which is the median of the predictions of LinearRegression models.","metadata":{}},{"cell_type":"markdown","source":"# Predict WEEK_NUM!","metadata":{}},{"cell_type":"code","source":"X_transform[\"WEEK_NUM_pred\"] = X_transform[enc.get_feature_names()].median(axis=1)\nnon_nan_rows = X_transform[\"WEEK_NUM_pred\"].notna()\nrmse_score = mean_squared_error(y[non_nan_rows], X_transform.loc[non_nan_rows, \"WEEK_NUM_pred\"], squared=False)\nprint(f\"Nan ratio: {(len(y)-sum(non_nan_rows))/len(y)}\")\nprint(f\"{rmse_score=}\")\nX_transform.tail()[\"WEEK_NUM_pred\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For training data, RMSE of predicted WEEK_NUM is 0.69! Not bad!\n\n... I thought at first.\n\nBut once you submit using the WEEK_NUM predictions, it does nothing to the score (Probably predictions are all nan or constant for test dataset). Afterward, I tried to predict WEEK_NUM with other features in vain. \n\nActually, one of the hosts said some features were transformed in test dataset, so I believe that's why this straightforward attempt fails.","metadata":{}},{"cell_type":"markdown","source":"# Save data","metadata":{}},{"cell_type":"code","source":"joblib.dump(list(df_date.columns), \"df_date_columns.joblib\")\njoblib.dump(enc, \"lr_pred_encoder.joblib\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}