{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Проделанная работа\n\n## Данные\nПопробовал 2 разных варианта аггрегации данных из предложенных таблиц:\n- В первом используется небольшое количество таблиц из csv файлов, игнорируется большинство таблиц credit_bureau, суммарно получается ~40 признаков, на первом запуске уже получались приличные скоры\n- Во втором варианте аггрегируются большинство предложенных таблиц из parquet-файлов, суммарно выходит ~400 признаков (всё ещё достаточно разреженных). Но приходится обрезать достаточно большую долю итоговой обучающей выборки из-за переполнения RAM. Скоры с чистого запуска получались намного лучше, но т.к. при сабмите используется другая тестовая выборка, происходит повторное переполнение памяти. Пробовал уменьшать долю обучающей выборки вплоть до 0.1 от изначальной, но всё равно сталкивался с Memory Limit, так что пришлось отказаться от этого подхода и перейти к работе с первым.\n\n## EDA\n- Таргет сильно несбалансирован, использовал classes_weights - 0.968 и 0.032 согласно соотношению классов в обучающей выборке\n- Корреляции между признаками датасетов - строил хитмапы для обоих датасетов, никакой явной связи отдельных признаков с таргетом нет\n- Проверял долю пропусков в данных - некоторые признаки даже в небольшом датасете отсутствуют в 90+% случаев. Простые способы избавления от строк с большим количеством пропусков не давали улучшения результатов\n- Пробовал рассматривать зависимости данных по временной шкале, но в последствие узнал, что в тестовой выборке все данные \"из будущего\", так что все эти планы свернул. Тем не менее, некоторые признаки сильно завязаны на временном промежутке, я думаю, что из подобных зависимостей что-то можно было получить.\n- Строил распределения признаков небольшого датасета (38 действительнозначных) - в абсолютном большинстве случаев распределения похожи на лог-нормальные, поэтому решил: отнормировать признаки с помощью min-max-scaling'а и прологарифмировать их. По предоставленному сравнению новых распредлений видно, что большинство распределений стали ближе к нормальным. Это позволило добиться небольших улучшений (+0.5% на валидации) в качестве по финальной метрике.\n\n## Модели\n- В качестве бейзлайна использовался lightGBM с подобранными руками параметрами - тк модель быстро переобучается, процесс обучения занимает достаточно мало времени и перебирать руками параметры достаточно удобно\n- Сначала сравнивал с CatboostClassifier, но он стабильно выдавал худшее качество. Проверял корректность работы с пропусками в данных, но это не изменило предпочтений в сторону lightGBM\n- Перешёл к XGBoost, на нём эксперементировал с подбором параметров больше всего, тк метрики были всё ещё хуже lightGBM, но значительно превосходили CatBoost.\n- Пробовал подбирать параметры XGBoost через Оптуну, но поиск параметров по разреженной сетке не дал улучшений, подобранные руками параметры оказались даже лучше.\n- В целом, XGBoost выглядит как overkill для нашей задачи, особенно для небольшого датасета. Бустинги очень быстро сталкиваются с переобучением и их приходится крайне жёстко регуляризовывать, из-за чего не удаётся раскрыть потенциал более сложных моделей. Судя по всему, бэггинг lightGBM эффективнее проявил себя с точки зрения регуляризации.\n\nИтоговый сабмит вышел плохим, тк потратил очень много сабмитов на попытки завести большой датасет на тестовых данных. В итоге сабмиты просроченного дня подходили к концу и решил заслать точно работающее решение на чистом ligthGBM :(","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport lightgbm as lgb\n\nfrom glob import glob\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score \n\ndataPath = \"/kaggle/input/home-credit-credit-risk-model-stability/\"","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:16.674649Z","iopub.execute_input":"2024-03-21T23:59:16.675470Z","iopub.status.idle":"2024-03-21T23:59:22.050135Z","shell.execute_reply.started":"2024-03-21T23:59:16.675435Z","shell.execute_reply":"2024-03-21T23:59:22.049291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class Pipeline:\n#     @staticmethod\n#     def set_table_dtypes(df):\n#         for col in df.columns:\n#             if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n#                 df = df.with_columns(pl.col(col).cast(pl.Int32))\n#             elif col in [\"date_decision\"]:\n#                 df = df.with_columns(pl.col(col).cast(pl.Date))\n#             elif col[-1] in (\"P\", \"A\"):\n#                 df = df.with_columns(pl.col(col).cast(pl.Float64))\n#             elif col[-1] in (\"M\",):\n#                 df = df.with_columns(pl.col(col).cast(pl.String))\n#             elif col[-1] in (\"D\",):\n#                 df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n#         return df\n    \n#     @staticmethod\n#     def handle_dates(df):\n#         for col in df.columns:\n#             if col[-1] in (\"D\",):\n#                 df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n#                 df = df.with_columns(pl.col(col).dt.total_days())\n#                 df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n#         df = df.drop(\"date_decision\", \"MONTH\")\n\n#         return df\n    \n#     @staticmethod\n#     def filter_cols(df):\n#         for col in df.columns:\n#             if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n#                 isnull = df[col].is_null().mean()\n\n#                 if isnull > 0.95:\n#                     df = df.drop(col)\n\n#         for col in df.columns:\n#             if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n#                 freq = df[col].n_unique()\n\n#                 if (freq == 1) | (freq > 200):\n#                     df = df.drop(col)\n\n#         return df","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.051701Z","iopub.execute_input":"2024-03-21T23:59:22.052017Z","iopub.status.idle":"2024-03-21T23:59:22.057685Z","shell.execute_reply.started":"2024-03-21T23:59:22.051990Z","shell.execute_reply":"2024-03-21T23:59:22.056631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class Aggregator:\n#     @staticmethod\n#     def num_expr(df):\n#         cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n#         expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n#         return expr_max\n\n#     @staticmethod\n#     def date_expr(df):\n#         cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n#         expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n#         return expr_max\n\n#     @staticmethod\n#     def str_expr(df):\n#         cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n#         expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n#         return expr_max\n\n#     @staticmethod\n#     def other_expr(df):\n#         cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n#         expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n#         return expr_max\n    \n#     @staticmethod\n#     def count_expr(df):\n#         cols = [col for col in df.columns if \"num_group\" in col]\n\n#         expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n#         return expr_max\n\n#     @staticmethod\n#     def get_exprs(df):\n#         exprs = Aggregator.num_expr(df) + \\\n#                 Aggregator.date_expr(df) + \\\n#                 Aggregator.str_expr(df) + \\\n#                 Aggregator.other_expr(df) + \\\n#                 Aggregator.count_expr(df)\n\n#         return exprs\n","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.058828Z","iopub.execute_input":"2024-03-21T23:59:22.059124Z","iopub.status.idle":"2024-03-21T23:59:22.080895Z","shell.execute_reply.started":"2024-03-21T23:59:22.059099Z","shell.execute_reply":"2024-03-21T23:59:22.079973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def read_file(path, depth=None):\n#     df = pl.read_parquet(path)\n#     df = df.pipe(Pipeline.set_table_dtypes)\n    \n#     if depth in [1, 2]:\n#         df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n#     return df\n\n# def read_files(regex_path, depth=None):\n#     chunks = []\n#     for path in glob(str(regex_path)):\n#         df = pl.read_parquet(path)\n#         df = df.pipe(Pipeline.set_table_dtypes)\n        \n#         if depth in [1, 2]:\n#             df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n#         chunks.append(df)\n        \n#     df = pl.concat(chunks, how=\"vertical_relaxed\")\n#     df = df.unique(subset=[\"case_id\"])\n    \n#     return df","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.083190Z","iopub.execute_input":"2024-03-21T23:59:22.083986Z","iopub.status.idle":"2024-03-21T23:59:22.094437Z","shell.execute_reply.started":"2024-03-21T23:59:22.083958Z","shell.execute_reply":"2024-03-21T23:59:22.093571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def feature_eng(df_base, depth_0, depth_1, depth_2):\n#     df_base = (\n#         df_base\n#         .with_columns(\n#             month_decision = pl.col(\"date_decision\").dt.month(),\n#             weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n#         )\n#     )\n        \n#     for i, df in enumerate(depth_0 + depth_1 + depth_2):\n#         df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n#     df_base = df_base.pipe(Pipeline.handle_dates)\n    \n#     return df_base","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.095539Z","iopub.execute_input":"2024-03-21T23:59:22.095875Z","iopub.status.idle":"2024-03-21T23:59:22.108133Z","shell.execute_reply.started":"2024-03-21T23:59:22.095842Z","shell.execute_reply":"2024-03-21T23:59:22.107296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def to_pandas(df_data, cat_cols=None):\n#     df_data = df_data.to_pandas()\n    \n#     if cat_cols is None:\n#         cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n#     df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n#     return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.109169Z","iopub.execute_input":"2024-03-21T23:59:22.109417Z","iopub.status.idle":"2024-03-21T23:59:22.118365Z","shell.execute_reply.started":"2024-03-21T23:59:22.109394Z","shell.execute_reply":"2024-03-21T23:59:22.117560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\n# TRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\n# TEST_DIR        = ROOT / \"parquet_files\" / \"test\"\n\n# train_data_store = {\n#     \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n#     \"depth_0\": [\n#         read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n#         read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n#     ],\n#     \"depth_1\": [\n#         read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n#         read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n#         read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n#     ],\n#     \"depth_2\": [\n#         read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n#         read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n#     ]\n# }\n\n# test_data_store = {\n#     \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n#     \"depth_0\": [\n#         read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n#         read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n#     ],\n#     \"depth_1\": [\n#         read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n#         read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n#         read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n#         read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n#     ],\n#     \"depth_2\": [\n#         read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n#         read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n#     ]\n# }\n\n# df_train = feature_eng(**train_data_store)\n# # df_test = feature_eng(**test_data_store)\n\n# print(\"train data shape:\\t\", df_train.shape)\n# # print(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.119861Z","iopub.execute_input":"2024-03-21T23:59:22.120181Z","iopub.status.idle":"2024-03-21T23:59:22.130912Z","shell.execute_reply.started":"2024-03-21T23:59:22.120150Z","shell.execute_reply":"2024-03-21T23:59:22.130084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train = df_train.pipe(Pipeline.filter_cols)\n# # df_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\n# print(\"train data shape:\\t\", df_train.shape)\n# # print(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.131981Z","iopub.execute_input":"2024-03-21T23:59:22.132988Z","iopub.status.idle":"2024-03-21T23:59:22.144273Z","shell.execute_reply.started":"2024-03-21T23:59:22.132955Z","shell.execute_reply":"2024-03-21T23:59:22.143427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_columns = [col for col in df_train.columns if col != \"target\"]","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.145494Z","iopub.execute_input":"2024-03-21T23:59:22.145837Z","iopub.status.idle":"2024-03-21T23:59:22.156497Z","shell.execute_reply.started":"2024-03-21T23:59:22.145805Z","shell.execute_reply":"2024-03-21T23:59:22.155766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import sys\n\n# # These are the usual ipython objects, including this one you are creating\n# ipython_vars = ['In', 'Out', 'exit', 'quit', 'get_ipython', 'ipython_vars']\n\n# # Get a sorted list of the objects and their sizes\n# sorted([(x, sys.getsizeof(globals().get(x))) for x in dir() if not x.startswith('_') and x not in sys.modules and x not in ipython_vars], key=lambda x: x[1], reverse=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.160716Z","iopub.execute_input":"2024-03-21T23:59:22.161029Z","iopub.status.idle":"2024-03-21T23:59:22.167374Z","shell.execute_reply.started":"2024-03-21T23:59:22.160997Z","shell.execute_reply":"2024-03-21T23:59:22.166495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train = df_train[:250000]","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.168324Z","iopub.execute_input":"2024-03-21T23:59:22.168570Z","iopub.status.idle":"2024-03-21T23:59:22.176800Z","shell.execute_reply.started":"2024-03-21T23:59:22.168547Z","shell.execute_reply":"2024-03-21T23:59:22.175874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train, cat_cols = to_pandas(df_train)\n# # df_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.177873Z","iopub.execute_input":"2024-03-21T23:59:22.178957Z","iopub.status.idle":"2024-03-21T23:59:22.186045Z","shell.execute_reply.started":"2024-03-21T23:59:22.178930Z","shell.execute_reply":"2024-03-21T23:59:22.185209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import sys\n\n# # These are the usual ipython objects, including this one you are creating\n# ipython_vars = ['In', 'Out', 'exit', 'quit', 'get_ipython', 'ipython_vars']\n\n# # Get a sorted list of the objects and their sizes\n# sorted([(x, sys.getsizeof(globals().get(x))) for x in dir() if not x.startswith('_') and x not in sys.modules and x not in ipython_vars], key=lambda x: x[1], reverse=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.187246Z","iopub.execute_input":"2024-03-21T23:59:22.187572Z","iopub.status.idle":"2024-03-21T23:59:22.195578Z","shell.execute_reply.started":"2024-03-21T23:59:22.187541Z","shell.execute_reply":"2024-03-21T23:59:22.194717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# (df_train.dtypes == 'category').sum()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.196668Z","iopub.execute_input":"2024-03-21T23:59:22.196996Z","iopub.status.idle":"2024-03-21T23:59:22.205307Z","shell.execute_reply.started":"2024-03-21T23:59:22.196963Z","shell.execute_reply":"2024-03-21T23:59:22.204561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = df_train.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"])\n# y = df_train[\"target\"]\n\n# indices = np.arange(X.shape[0])\n# train_indices, valid_indices = train_test_split(indices, train_size=0.8, random_state=1)\n# valid_indices, test_indices = train_test_split(valid_indices, train_size=0.6, random_state=1)\n\n# X_train = X.iloc[train_indices]\n# y_train = y.iloc[train_indices]\n\n# X_valid = X.iloc[valid_indices] # .copy()\n# y_valid = y.iloc[valid_indices] # .copy()\n\n# X_test = X.iloc[test_indices] # .copy()\n# y_test = y.iloc[test_indices] # .copy()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.206445Z","iopub.execute_input":"2024-03-21T23:59:22.207048Z","iopub.status.idle":"2024-03-21T23:59:22.215127Z","shell.execute_reply.started":"2024-03-21T23:59:22.207016Z","shell.execute_reply":"2024-03-21T23:59:22.214464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(f\"Train: {X_train.shape}\")\n# print(f\"Valid: {X_valid.shape}\")\n# print(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.216251Z","iopub.execute_input":"2024-03-21T23:59:22.216862Z","iopub.status.idle":"2024-03-21T23:59:22.224848Z","shell.execute_reply.started":"2024-03-21T23:59:22.216829Z","shell.execute_reply":"2024-03-21T23:59:22.224077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.225776Z","iopub.execute_input":"2024-03-21T23:59:22.226002Z","iopub.status.idle":"2024-03-21T23:59:22.235083Z","shell.execute_reply.started":"2024-03-21T23:59:22.225980Z","shell.execute_reply":"2024-03-21T23:59:22.234348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_table_dtypes(df: pl.DataFrame) -> pl.DataFrame:\n    # implement here all desired dtypes for tables\n    # the following is just an example\n    for col in df.columns:\n        # last letter of column name will help you determine the type\n        if col[-1] in (\"P\", \"A\"):\n            df = df.with_columns(pl.col(col).cast(pl.Float64).alias(col))\n\n    return df\n\ndef convert_strings(df: pd.DataFrame) -> pd.DataFrame:\n    for col in df.columns:  \n        if df[col].dtype.name in ['object', 'string']:\n            df[col] = df[col].astype(\"string\").astype('category')\n            current_categories = df[col].cat.categories\n            new_categories = current_categories.to_list() + [\"Unknown\"]\n            new_dtype = pd.CategoricalDtype(categories=new_categories, ordered=True)\n            df[col] = df[col].astype(new_dtype)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.236069Z","iopub.execute_input":"2024-03-21T23:59:22.236296Z","iopub.status.idle":"2024-03-21T23:59:22.245613Z","shell.execute_reply.started":"2024-03-21T23:59:22.236274Z","shell.execute_reply":"2024-03-21T23:59:22.244870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_basetable = pl.read_csv(dataPath + \"csv_files/train/train_base.csv\")\ntrain_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/train/train_static_0_1.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntrain_static_cb = pl.read_csv(dataPath + \"csv_files/train/train_static_cb_0.csv\").pipe(set_table_dtypes)\ntrain_person_1 = pl.read_csv(dataPath + \"csv_files/train/train_person_1.csv\").pipe(set_table_dtypes) \ntrain_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/train/train_credit_bureau_b_2.csv\").pipe(set_table_dtypes) \n# added a2\n# train_credit_bureau_a_2_list = []\n# for i in range(3):\n#     train_credit_bureau_a_2_list.append(\n#         pl.read_csv(dataPath + f\"csv_files/train/train_credit_bureau_a_2_{i}.csv\").pipe(set_table_dtypes) \n#     )","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:22.246621Z","iopub.execute_input":"2024-03-21T23:59:22.246967Z","iopub.status.idle":"2024-03-21T23:59:36.064027Z","shell.execute_reply.started":"2024-03-21T23:59:22.246942Z","shell.execute_reply":"2024-03-21T23:59:36.063120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_basetable = pl.read_csv(dataPath + \"csv_files/test/test_base.csv\")\ntest_static = pl.concat(\n    [\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_0.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_1.csv\").pipe(set_table_dtypes),\n        pl.read_csv(dataPath + \"csv_files/test/test_static_0_2.csv\").pipe(set_table_dtypes),\n    ],\n    how=\"vertical_relaxed\",\n)\ntest_static_cb = pl.read_csv(dataPath + \"csv_files/test/test_static_cb_0.csv\").pipe(set_table_dtypes)\ntest_person_1 = pl.read_csv(dataPath + \"csv_files/test/test_person_1.csv\").pipe(set_table_dtypes) \ntest_credit_bureau_b_2 = pl.read_csv(dataPath + \"csv_files/test/test_credit_bureau_b_2.csv\").pipe(set_table_dtypes) \n\n# added a2\n# test_credit_bureau_a_2_list = []\n# for i in range(3):\n#     test_credit_bureau_a_2_list.append(\n#         pl.read_csv(dataPath + f\"csv_files/test/test_credit_bureau_a_2_{i}.csv\").pipe(set_table_dtypes) \n#     )","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:36.065179Z","iopub.execute_input":"2024-03-21T23:59:36.066995Z","iopub.status.idle":"2024-03-21T23:59:36.139979Z","shell.execute_reply.started":"2024-03-21T23:59:36.066967Z","shell.execute_reply":"2024-03-21T23:59:36.139106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering\n\nIn this part, we can see a simple example of joining tables via `case_id`. Here the loading and joining is done with polars library. Polars library is blazingly fast and has much smaller memory footprint than pandas. ","metadata":{}},{"cell_type":"code","source":"# We need to use aggregation functions in tables with depth > 1, so tables that contain num_group1 column or \n# also num_group2 column.\ntrain_person_1_feats_1 = train_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\n# Here num_group1=0 has special meaning, it is the person who applied for the loan.\ntrain_person_1_feats_2 = train_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\n# Here we have num_goup1 and num_group2, so we need to aggregate again.\ntrain_credit_bureau_b_2_feats = train_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# train_credit_bureau_a_2_feats = [train_credit_bureau_a_2.group_by(\"case_id\").agg(\n#     pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n#     (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n# ) for train_credit_bureau_a_2 in train_credit_bureau_a_2_list]\n\n# We will process in this examples only A-type and M-type columns, so we need to select them.\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cols.append(col)\nprint(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"M\"):\n        selected_static_cb_cols.append(col)\nprint(selected_static_cb_cols)\n\n# Join all tables together.\ndata = train_basetable.join(\n    train_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    train_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    train_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n) #.join(\n#     train_credit_bureau_a_2_feats[0], how=\"left\", on=\"case_id\"\n# ).join(\n#     train_credit_bureau_a_2_feats[1], how=\"left\", on=\"case_id\"\n# ).join(\n#     train_credit_bureau_a_2_feats[2], how=\"left\", on=\"case_id\"\n# )","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:36.141303Z","iopub.execute_input":"2024-03-21T23:59:36.141880Z","iopub.status.idle":"2024-03-21T23:59:37.533425Z","shell.execute_reply.started":"2024-03-21T23:59:36.141846Z","shell.execute_reply":"2024-03-21T23:59:37.532406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_person_1_feats_1 = test_person_1.group_by(\"case_id\").agg(\n    pl.col(\"mainoccupationinc_384A\").max().alias(\"mainoccupationinc_384A_max\"),\n    (pl.col(\"incometype_1044T\") == \"SELFEMPLOYED\").max().alias(\"mainoccupationinc_384A_any_selfemployed\")\n)\n\ntest_person_1_feats_2 = test_person_1.select([\"case_id\", \"num_group1\", \"housetype_905L\"]).filter(\n    pl.col(\"num_group1\") == 0\n).drop(\"num_group1\").rename({\"housetype_905L\": \"person_housetype\"})\n\ntest_credit_bureau_b_2_feats = test_credit_bureau_b_2.group_by(\"case_id\").agg(\n    pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n    (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n)\n\n# test_credit_bureau_a_2_feats = test_credit_bureau_a_2.group_by(\"case_id\").agg(\n#     pl.col(\"pmts_pmtsoverdue_635A\").max().alias(\"pmts_pmtsoverdue_635A_max\"),\n#     (pl.col(\"pmts_dpdvalue_108P\") > 31).max().alias(\"pmts_dpdvalue_108P_over31\")\n# )\n\ndata_submission = test_basetable.join(\n    test_static.select([\"case_id\"]+selected_static_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_static_cb.select([\"case_id\"]+selected_static_cb_cols), how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_1, how=\"left\", on=\"case_id\"\n).join(\n    test_person_1_feats_2, how=\"left\", on=\"case_id\"\n).join(\n    test_credit_bureau_b_2_feats, how=\"left\", on=\"case_id\"\n) # ).join(\n#     test_credit_bureau_a_2_feats, how=\"left\", on=\"case_id\"\n# )","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:37.534668Z","iopub.execute_input":"2024-03-21T23:59:37.534991Z","iopub.status.idle":"2024-03-21T23:59:37.546689Z","shell.execute_reply.started":"2024-03-21T23:59:37.534957Z","shell.execute_reply":"2024-03-21T23:59:37.545904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"case_ids = data[\"case_id\"].unique().shuffle(seed=1)\ncase_ids_train, case_ids_test = train_test_split(case_ids, train_size=0.99, random_state=1)\ncase_ids_valid, case_ids_test = train_test_split(case_ids_test, train_size=0.5, random_state=1)\n\ncols_pred = []\nfor col in data.columns:\n    if col[-1].isupper() and col[:-1].islower():\n        cols_pred.append(col)\n\nprint(cols_pred)\n\ndef from_polars_to_pandas(case_ids: pl.DataFrame) -> pl.DataFrame:\n    return (\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[[\"case_id\", \"WEEK_NUM\", \"target\"]].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[cols_pred].to_pandas(),\n        data.filter(pl.col(\"case_id\").is_in(case_ids))[\"target\"].to_pandas()\n    )\n\nbase_train, X_train, y_train = from_polars_to_pandas(case_ids_train)\nbase_valid, X_valid, y_valid = from_polars_to_pandas(case_ids_valid)\nbase_test, X_test, y_test = from_polars_to_pandas(case_ids_test)\n\nfor df in [X_train, X_valid, X_test]:\n    df = convert_strings(df)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:37.547761Z","iopub.execute_input":"2024-03-21T23:59:37.548108Z","iopub.status.idle":"2024-03-21T23:59:44.431795Z","shell.execute_reply.started":"2024-03-21T23:59:37.548081Z","shell.execute_reply":"2024-03-21T23:59:44.430990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_train.WEEK_NUM","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:44.433159Z","iopub.execute_input":"2024-03-21T23:59:44.433446Z","iopub.status.idle":"2024-03-21T23:59:44.441901Z","shell.execute_reply.started":"2024-03-21T23:59:44.433420Z","shell.execute_reply":"2024-03-21T23:59:44.440957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.plot.hist(title='target distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:44.443110Z","iopub.execute_input":"2024-03-21T23:59:44.443409Z","iopub.status.idle":"2024-03-21T23:59:44.786224Z","shell.execute_reply.started":"2024-03-21T23:59:44.443372Z","shell.execute_reply":"2024-03-21T23:59:44.785319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# WE'LL NEED TARGET WEIGHT SCALING - 0.032 and 0.968 \n(y_train == 0).sum() / y_train.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:44.787299Z","iopub.execute_input":"2024-03-21T23:59:44.787568Z","iopub.status.idle":"2024-03-21T23:59:44.795492Z","shell.execute_reply.started":"2024-03-21T23:59:44.787542Z","shell.execute_reply":"2024-03-21T23:59:44.794596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train: {X_train.shape}\")\nprint(f\"Valid: {X_valid.shape}\")\nprint(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:44.796725Z","iopub.execute_input":"2024-03-21T23:59:44.797036Z","iopub.status.idle":"2024-03-21T23:59:44.802526Z","shell.execute_reply.started":"2024-03-21T23:59:44.797009Z","shell.execute_reply":"2024-03-21T23:59:44.801657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:44.809275Z","iopub.execute_input":"2024-03-21T23:59:44.809549Z","iopub.status.idle":"2024-03-21T23:59:44.817299Z","shell.execute_reply.started":"2024-03-21T23:59:44.809525Z","shell.execute_reply":"2024-03-21T23:59:44.816518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = X_train.copy()\ntrain_data['target'] = y_train","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:44.818206Z","iopub.execute_input":"2024-03-21T23:59:44.818436Z","iopub.status.idle":"2024-03-21T23:59:44.985637Z","shell.execute_reply.started":"2024-03-21T23:59:44.818415Z","shell.execute_reply":"2024-03-21T23:59:44.984887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IT WAS A HUGE HEATMAP FOR THE LARGE DATASET\n\n# plt.figure(figsize=(15, 12))\n# sns.heatmap(train_data.corr(numeric_only=True))\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:44.986602Z","iopub.execute_input":"2024-03-21T23:59:44.986878Z","iopub.status.idle":"2024-03-21T23:59:44.990780Z","shell.execute_reply.started":"2024-03-21T23:59:44.986853Z","shell.execute_reply":"2024-03-21T23:59:44.989959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# On the small dataset \n# Target is the last column in both heatmaps and is almost not-correlated with other features\n\ntrain_data = X_train.copy()\ntrain_data['target'] = y_train\n\nplt.figure(figsize=(15, 12))\nsns.heatmap(train_data.corr(numeric_only=True))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:44.992035Z","iopub.execute_input":"2024-03-21T23:59:44.992286Z","iopub.status.idle":"2024-03-21T23:59:49.147304Z","shell.execute_reply.started":"2024-03-21T23:59:44.992262Z","shell.execute_reply":"2024-03-21T23:59:49.146350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(train_data[['annuity_780A', 'target']], x='annuity_780A', y='target')","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:49.148416Z","iopub.execute_input":"2024-03-21T23:59:49.148725Z","iopub.status.idle":"2024-03-21T23:59:54.365161Z","shell.execute_reply.started":"2024-03-21T23:59:49.148698Z","shell.execute_reply":"2024-03-21T23:59:54.364259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(train_data[['avglnamtstart24m_4525187A', 'target']], x='avglnamtstart24m_4525187A', y='target')","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:54.366458Z","iopub.execute_input":"2024-03-21T23:59:54.366851Z","iopub.status.idle":"2024-03-21T23:59:55.661971Z","shell.execute_reply.started":"2024-03-21T23:59:54.366814Z","shell.execute_reply":"2024-03-21T23:59:55.661052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.isna().sum(axis=0) / X_train.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:55.663232Z","iopub.execute_input":"2024-03-21T23:59:55.663597Z","iopub.status.idle":"2024-03-21T23:59:55.762115Z","shell.execute_reply.started":"2024-03-21T23:59:55.663560Z","shell.execute_reply":"2024-03-21T23:59:55.761220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:55.763302Z","iopub.execute_input":"2024-03-21T23:59:55.763662Z","iopub.status.idle":"2024-03-21T23:59:57.964430Z","shell.execute_reply.started":"2024-03-21T23:59:55.763625Z","shell.execute_reply":"2024-03-21T23:59:57.963435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape[1]","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:57.965724Z","iopub.execute_input":"2024-03-21T23:59:57.967017Z","iopub.status.idle":"2024-03-21T23:59:57.973039Z","shell.execute_reply.started":"2024-03-21T23:59:57.966987Z","shell.execute_reply":"2024-03-21T23:59:57.972119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FURTHER PLOTS ARE FOR THE SMALL DATASET\n# MOST LIKELY, DISTRIBUTIONS OF THE MAJORITY OF ADDED NUMERICAL FEATURES ARE ALSO LOG-NORMAL","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:57.975129Z","iopub.execute_input":"2024-03-21T23:59:57.975441Z","iopub.status.idle":"2024-03-21T23:59:57.981296Z","shell.execute_reply.started":"2024-03-21T23:59:57.975415Z","shell.execute_reply":"2024-03-21T23:59:57.980415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:57.982271Z","iopub.execute_input":"2024-03-21T23:59:57.982542Z","iopub.status.idle":"2024-03-21T23:59:57.989530Z","shell.execute_reply.started":"2024-03-21T23:59:57.982518Z","shell.execute_reply":"2024-03-21T23:59:57.988781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_rows, n_cols = 6, 6\nfig, axs = plt.subplots(n_rows, n_cols, figsize=(20, 16))\n\n# train_data.plot(kind='hist', subplots=True, layout=(6, 6), ax=ax)\nfor i in range(n_rows):\n    for j in range(n_cols):\n        sns.histplot(train_data.iloc[:, i * n_cols + j].dropna(), ax=axs[i][j])\n        axs[i][j].set_xticks([])\n        axs[i][j].set_yticks([])\n        axs[i][j].set_title(train_data.columns[i * n_cols + j])\n        axs[i][j].set_xlabel('')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-21T23:59:57.990551Z","iopub.execute_input":"2024-03-21T23:59:57.990880Z","iopub.status.idle":"2024-03-22T00:03:02.345042Z","shell.execute_reply.started":"2024-03-21T23:59:57.990854Z","shell.execute_reply":"2024-03-22T00:03:02.344072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_rows, n_cols = 6, 6\nfig, axs = plt.subplots(n_rows, n_cols, figsize=(20, 16))\n\n# train_data.plot(kind='hist', subplots=True, layout=(6, 6), ax=ax)\nfor i in range(n_rows):\n    for j in range(n_cols):\n        sns.histplot(train_data.iloc[:, i * n_cols + j].dropna(), ax=axs[i][j])\n        axs[i][j].set_xticks([])\n        axs[i][j].set_yticks([])\n        axs[i][j].set_xscale('log')\n        axs[i][j].set_title(train_data.columns[i * n_cols + j])\n        axs[i][j].set_xlabel('')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:20:26.127711Z","iopub.execute_input":"2024-03-22T00:20:26.128068Z","iopub.status.idle":"2024-03-22T00:23:57.317235Z","shell.execute_reply.started":"2024-03-22T00:20:26.128040Z","shell.execute_reply":"2024-03-22T00:23:57.316294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_indices = [i for i, x in enumerate(X_train.dtypes) if x == 'category']\nnum_indices = [i for i, x in enumerate(X_train.dtypes) if x == 'float64']\n\nnorm_train_data = train_data.copy()\ncolumn_mins = train_data.iloc[:, num_indices].min()\ncolumn_maxs = train_data.iloc[:, num_indices].max()\nnorm_train_data.iloc[:, num_indices] = (train_data.iloc[:, num_indices] - column_mins) / (column_maxs - column_mins) + 1","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:23:57.318828Z","iopub.execute_input":"2024-03-22T00:23:57.319167Z","iopub.status.idle":"2024-03-22T00:23:58.717264Z","shell.execute_reply.started":"2024-03-22T00:23:57.319137Z","shell.execute_reply":"2024-03-22T00:23:58.716439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_rows, n_cols = 6, 6\nfig, axs = plt.subplots(n_rows, n_cols, figsize=(20, 16))\n\n# train_data.plot(kind='hist', subplots=True, layout=(6, 6), ax=ax)\nfor i in range(n_rows):\n    for j in range(n_cols):\n        sns.histplot(norm_train_data.iloc[:, i * n_cols + j].dropna(), ax=axs[i][j])\n        axs[i][j].set_xticks([])\n        axs[i][j].set_yticks([])\n        axs[i][j].set_title(norm_train_data.columns[i * n_cols + j])\n        axs[i][j].set_xlabel('')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:23:58.718408Z","iopub.execute_input":"2024-03-22T00:23:58.718787Z","iopub.status.idle":"2024-03-22T00:27:09.909619Z","shell.execute_reply.started":"2024-03-22T00:23:58.718750Z","shell.execute_reply":"2024-03-22T00:27:09.908720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_rows, n_cols = 6, 6\nfig, axs = plt.subplots(n_rows, n_cols, figsize=(20, 16))\n\n# train_data.plot(kind='hist', subplots=True, layout=(6, 6), ax=ax)\nfor i in range(n_rows):\n    for j in range(n_cols):\n        sns.histplot(norm_train_data.iloc[:, i * n_cols + j].dropna(), ax=axs[i][j])\n        axs[i][j].set_xticks([])\n        axs[i][j].set_yticks([])\n        axs[i][j].set_xscale('log')\n        axs[i][j].set_title(norm_train_data.columns[i * n_cols + j])\n        axs[i][j].set_xlabel('')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:27:09.912204Z","iopub.execute_input":"2024-03-22T00:27:09.912509Z","iopub.status.idle":"2024-03-22T00:30:37.421991Z","shell.execute_reply.started":"2024-03-22T00:27:09.912480Z","shell.execute_reply":"2024-03-22T00:30:37.421057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.columns","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:30:37.423298Z","iopub.execute_input":"2024-03-22T00:30:37.423602Z","iopub.status.idle":"2024-03-22T00:30:37.430224Z","shell.execute_reply.started":"2024-03-22T00:30:37.423575Z","shell.execute_reply":"2024-03-22T00:30:37.429371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# log_scale_features = ['amtinstpaidbefduel24m_4187115A', 'annuity_780A', 'avginstallast24m_3658937A']","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:30:37.431587Z","iopub.execute_input":"2024-03-22T00:30:37.432231Z","iopub.status.idle":"2024-03-22T00:30:37.439341Z","shell.execute_reply.started":"2024-03-22T00:30:37.432197Z","shell.execute_reply":"2024-03-22T00:30:37.438483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"norm_train_data.iloc[:, num_indices] = norm_train_data.iloc[:, num_indices]\nfor i in num_indices:\n    norm_train_data.iloc[:, i] = np.log(norm_train_data.iloc[:, i])","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:20.588779Z","iopub.execute_input":"2024-03-22T00:31:20.589441Z","iopub.status.idle":"2024-03-22T00:31:21.040619Z","shell.execute_reply.started":"2024-03-22T00:31:20.589398Z","shell.execute_reply":"2024-03-22T00:31:21.039829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:21.121646Z","iopub.execute_input":"2024-03-22T00:31:21.121948Z","iopub.status.idle":"2024-03-22T00:31:21.146591Z","shell.execute_reply.started":"2024-03-22T00:31:21.121922Z","shell.execute_reply":"2024-03-22T00:31:21.145769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"norm_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:22.280899Z","iopub.execute_input":"2024-03-22T00:31:22.281277Z","iopub.status.idle":"2024-03-22T00:31:22.307042Z","shell.execute_reply.started":"2024-03-22T00:31:22.281248Z","shell.execute_reply":"2024-03-22T00:31:22.306013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# As we want our models to perform to the future changes, we aim for the time-dependent robust predictions\n# Let's look at the feature dependencies on the row index (rows are time-sorted)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:23.285371Z","iopub.execute_input":"2024-03-22T00:31:23.286251Z","iopub.status.idle":"2024-03-22T00:31:23.290274Z","shell.execute_reply.started":"2024-03-22T00:31:23.286214Z","shell.execute_reply":"2024-03-22T00:31:23.289164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_rows, n_cols = 6, 6\nfig, axs = plt.subplots(n_rows, n_cols, figsize=(20, 16))\n\n# train_data.plot(kind='hist', subplots=True, layout=(6, 6), ax=ax)\nfor i in range(n_rows):\n    for j in range(n_cols):\n        axs[i][j].plot(train_data.iloc[:, i * n_cols + j].dropna())\n        axs[i][j].set_xticks([])\n        axs[i][j].set_yticks([])\n        axs[i][j].set_title(norm_train_data.columns[i * n_cols + j])\n        axs[i][j].set_xlabel('')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:24.136422Z","iopub.execute_input":"2024-03-22T00:31:24.137128Z","iopub.status.idle":"2024-03-22T00:31:34.430260Z","shell.execute_reply.started":"2024-03-22T00:31:24.137091Z","shell.execute_reply":"2024-03-22T00:31:34.429271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = X_test.copy()\ntest_data['target'] = y_test","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:34.432290Z","iopub.execute_input":"2024-03-22T00:31:34.432902Z","iopub.status.idle":"2024-03-22T00:31:34.440719Z","shell.execute_reply.started":"2024-03-22T00:31:34.432867Z","shell.execute_reply":"2024-03-22T00:31:34.439709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_rows, n_cols = 6, 6\nfig, axs = plt.subplots(n_rows, n_cols, figsize=(20, 16))\n\n# train_data.plot(kind='hist', subplots=True, layout=(6, 6), ax=ax)\nfor i in range(n_rows):\n    for j in range(n_cols):\n        axs[i][j].plot(test_data.iloc[:, i * n_cols + j].dropna())\n        axs[i][j].set_xticks([])\n        axs[i][j].set_yticks([])\n        axs[i][j].set_title(norm_train_data.columns[i * n_cols + j])\n        axs[i][j].set_xlabel('')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:34.441918Z","iopub.execute_input":"2024-03-22T00:31:34.442163Z","iopub.status.idle":"2024-03-22T00:31:36.409594Z","shell.execute_reply.started":"2024-03-22T00:31:34.442140Z","shell.execute_reply":"2024-03-22T00:31:36.408689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# WE HAVE THE SAME TIMELINE DEPENDENCIES, SO SHALL WE USE ROW INDICES AS FEATURES ???\n\n# WE MAY TRAIN ANOTHER MODEL TO PREDICT TIME_INDEX BY ROWS\n# THEREFORE, SIMILAR ROWS FROM THE SAME TIME PERIOD WILL GET SIMILAR TIME INDICES\n\n# HOWEVER, IN THE INITIAL DATA SOME TIME-REPRESENTING FEATURES EXIST, WE MAY ADD THEM INSTEAD OF SUCH PIPELINE\n# SO MOST-LIKELY, IT WOULD NOT GIVE US BETTER PERFORMANCE THAN SIMPLE FEATURE ADDITION","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:36.412103Z","iopub.execute_input":"2024-03-22T00:31:36.412703Z","iopub.status.idle":"2024-03-22T00:31:36.417155Z","shell.execute_reply.started":"2024-03-22T00:31:36.412669Z","shell.execute_reply":"2024-03-22T00:31:36.416191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# norm_train_data.iloc[:, num_indices] = norm_train_data.iloc[:, num_indices]\n\ncat_indices = [i for i, x in enumerate(X_train.dtypes) if x == 'category']\nnum_indices = [i for i, x in enumerate(X_train.dtypes) if x == 'float64']\n\nfor i in num_indices:\n    c = X_train.columns[i]\n    \n    X_train[c] = (X_train[c] - X_train[c].min()) / (X_train[c].max() - X_train[c].min()) + 1\n    X_test[c] = (X_test[c] - X_test[c].min()) / (X_test[c].max() - X_test[c].min()) + 1\n    X_valid[c] = (X_valid[c] - X_valid[c].min()) / (X_valid[c].max() - X_valid[c].min()) + 1\n\n    X_train[c] = np.log(X_train[c])\n    X_test[c] = np.log(X_test[c])\n    X_valid[c] = np.log(X_valid[c])\n\n# SHALL WE LEARN MINMAXSCALING ?","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:36.418364Z","iopub.execute_input":"2024-03-22T00:31:36.419070Z","iopub.status.idle":"2024-03-22T00:31:37.080472Z","shell.execute_reply.started":"2024-03-22T00:31:36.419036Z","shell.execute_reply":"2024-03-22T00:31:37.079708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.min(axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:37.081587Z","iopub.execute_input":"2024-03-22T00:31:37.082031Z","iopub.status.idle":"2024-03-22T00:31:37.211864Z","shell.execute_reply.started":"2024-03-22T00:31:37.081986Z","shell.execute_reply":"2024-03-22T00:31:37.210887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.isna().sum() / X_train.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:37.213046Z","iopub.execute_input":"2024-03-22T00:31:37.213305Z","iopub.status.idle":"2024-03-22T00:31:37.309351Z","shell.execute_reply.started":"2024-03-22T00:31:37.213282Z","shell.execute_reply":"2024-03-22T00:31:37.308509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, column_name in enumerate(X_train.columns):\n    if i in cat_indices:\n        X_train[column_name] = X_train[column_name].fillna(X_train[column_name].min(axis=0))\n        X_valid[column_name] = X_valid[column_name].fillna(X_valid[column_name].min(axis=0))\n        X_test[column_name] = X_test[column_name].fillna(X_test[column_name].min(axis=0))\n    else:\n        X_train[column_name] = X_train[column_name].fillna(X_train[column_name].min(axis=0) - 1)\n        X_valid[column_name] = X_valid[column_name].fillna(X_valid[column_name].min(axis=0) - 1)\n        X_test[column_name] = X_test[column_name].fillna(X_test[column_name].min(axis=0) - 1)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T00:31:37.310329Z","iopub.execute_input":"2024-03-22T00:31:37.310625Z","iopub.status.idle":"2024-03-22T00:31:37.601487Z","shell.execute_reply.started":"2024-03-22T00:31:37.310600Z","shell.execute_reply":"2024-03-22T00:31:37.600479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training LightGBM\n\nMinimal example of LightGBM training is shown below.","metadata":{}},{"cell_type":"code","source":"# TRAINED ON THE SMALL DATASET:\n\nlgb_train = lgb.Dataset(X_train, label=y_train)\nlgb_valid = lgb.Dataset(X_valid, label=y_valid, reference=lgb_train)\n\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"max_depth\": 3,\n    \"num_leaves\": 31,\n    \"learning_rate\": 0.05,\n    \"feature_fraction\": 0.9,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"n_estimators\": 1000,\n    \"verbose\": -1,\n}\n\ngbm = lgb.train(\n    params,\n    lgb_train,\n    valid_sets=lgb_valid,\n    callbacks=[lgb.log_evaluation(50), lgb.early_stopping(10)]\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T19:19:42.348625Z","iopub.execute_input":"2024-03-18T19:19:42.348961Z","iopub.status.idle":"2024-03-18T19:20:00.547950Z","shell.execute_reply.started":"2024-03-18T19:19:42.348930Z","shell.execute_reply":"2024-03-18T19:20:00.547215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ON THE HUGE ONE\n\n# lgb_train = lgb.Dataset(X_train, label=y_train)\n# lgb_valid = lgb.Dataset(X_valid, label=y_valid, reference=lgb_train)\n\n# params = {\n#     \"boosting_type\": \"gbdt\",\n#     \"objective\": \"binary\",\n#     \"metric\": \"auc\",\n#     \"max_depth\": 4,\n# #     \"num_leaves\": 31,\n#     \"learning_rate\": 0.05,\n#     \"colsample_bytree\": 0.8, \n#     \"colsample_bynode\": 0.8,\n# #     \"feature_fraction\": 0.9,\n# #     \"bagging_fraction\": 0.8,\n# #     \"bagging_freq\": 5,\n#     \"n_estimators\": 1000,\n#     \"verbose\": -1,\n#     \"device\": \"gpu\"\n# }\n\n# gbm = lgb.train(\n#     params,\n#     lgb_train,\n#     valid_sets=lgb_valid,\n#     callbacks=[lgb.log_evaluation(50), lgb.early_stopping(10)]\n# )\n\n# params = {\n#     \"boosting_type\": \"gbdt\",\n#     \"objective\": \"binary\",\n#     \"metric\": \"auc\",\n#     \"max_depth\": 8,\n#     \"learning_rate\": 0.05,\n#     \"n_estimators\": 1000,\n#     \"colsample_bytree\": 0.8, \n#     \"colsample_bynode\": 0.8,\n#     \"verbose\": -1,\n#     \"random_state\": 42,\n#     \"device\": \"gpu\",\n# }\n\n# model = lgb.LGBMClassifier(**params)\n# model.fit(\n#     X_train, y_train,\n#     eval_set=[(X_valid, y_valid)],\n#     callbacks=[lgb.log_evaluation(100), lgb.early_stopping(100)]\n# )","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:32:45.948487Z","iopub.execute_input":"2024-03-17T21:32:45.948864Z","iopub.status.idle":"2024-03-17T21:33:17.511934Z","shell.execute_reply.started":"2024-03-17T21:32:45.948833Z","shell.execute_reply":"2024-03-17T21:33:17.511095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical_features_indices = [i for i, x in enumerate(X_train.dtypes) if x == 'category']\n# print(categorical_features_indices)\n\n# for i in categorical_features_indices:\n#     for X in [X_train, X_test, X_valid]:\n#         X.iloc[:, i] = X.iloc[:, i].cat.add_categories('').fillna('')","metadata":{"execution":{"iopub.status.busy":"2024-03-16T14:14:27.322533Z","iopub.execute_input":"2024-03-16T14:14:27.324473Z","iopub.status.idle":"2024-03-16T14:14:27.372293Z","shell.execute_reply.started":"2024-03-16T14:14:27.324442Z","shell.execute_reply":"2024-03-16T14:14:27.371152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from catboost import CatBoostClassifier, Pool, metrics, cv\n\n# catboost_model = CatBoostClassifier(\n#     custom_loss=[metrics.CrossEntropy()],\n#     random_seed=42,\n#     logging_level='Silent',\n#     nan_mode='Min'\n# )\n\n# catboost_model.fit(\n#     X_train, y_train,\n#     cat_features=categorical_features_indices,\n#     eval_set=(X_valid, y_valid),\n#     logging_level='Verbose',  # you can uncomment this for text output\n#     plot=True\n# )","metadata":{"execution":{"iopub.status.busy":"2024-03-16T14:14:27.373859Z","iopub.execute_input":"2024-03-16T14:14:27.374257Z","iopub.status.idle":"2024-03-16T14:42:58.269416Z","shell.execute_reply.started":"2024-03-16T14:14:27.374220Z","shell.execute_reply":"2024-03-16T14:42:58.268511Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from catboost import CatBoostClassifier, Pool, metrics, cv\n\n# train_pool = Pool(\n#     X_train,\n#     y_train,\n#     cat_features=categorical_features_indices,\n#     weight=np.where(y_train, 0.968, 0.032)\n# )\n# valid_pool = Pool(\n#     X_valid,\n#     y_valid,\n#     cat_features=categorical_features_indices,\n#     weight=np.where(y_valid, 0.968, 0.032)\n# )\n# test_pool = Pool(\n#     X_test,\n#     y_test,\n#     cat_features=categorical_features_indices,\n#     weight=np.where(y_test, 0.968, 0.032)\n# )\n\n# catboost_model2 = CatBoostClassifier(\n#     custom_loss=[metrics.CrossEntropy()],\n#     random_seed=42,\n#     logging_level='Silent',\n#     nan_mode='Min',\n#     iterations=100\n# )\n\n# catboost_model2.fit(\n#     train_pool,\n#     eval_set=valid_pool,\n#     logging_level='Verbose',  # you can uncomment this for text output\n#     plot=True\n# )","metadata":{"execution":{"iopub.status.busy":"2024-03-16T15:07:46.224228Z","iopub.execute_input":"2024-03-16T15:07:46.224604Z","iopub.status.idle":"2024-03-16T15:09:01.329654Z","shell.execute_reply.started":"2024-03-16T15:07:46.224575Z","shell.execute_reply":"2024-03-16T15:09:01.328738Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import xgboost as xgb\n# from sklearn.preprocessing import StandardScaler\n# from sklearn.utils import class_weight\n\n# classes_weights = class_weight.compute_sample_weight(\n#     class_weight='balanced',\n#     y=y_train\n# )\n\n# # scaler = StandardScaler()\n# # X_train_scaled = scaler.fit_transform(X_train)\n\n# # xgb_params = {\n# #     'max_depth': 9,\n# #     'eta': 1,\n# #     'objective': 'binary:logistic',\n# #     'eval_metric': 'auc',\n# #     device='cuda'\n# # }\n\n# # dtrain = xgb.DMatrix(X_, label=label)\n\n# xgbmodel = xgb.XGBClassifier(\n#     n_estimators=600, max_depth=4, learning_rate=0.05, objective='binary:logistic',\n#     reg_alpha=0.0, reg_lambda=10.0, max_leaves=10, min_child_weight=30.0, gamma=0.1,\n#     tree_method=\"hist\", enable_categorical=True, eval_metric='auc', device=\"cuda\"\n# )\n\n# # WITHOUT CLASSES_WEIGHTS IT WORKS A BIT BETTER\n# xgbmodel = xgbmodel.fit(X_train, y_train, eval_set=[(X_valid, y_valid)]) #, sample_weight=classes_weights)","metadata":{"execution":{"iopub.status.busy":"2024-03-16T19:09:05.814055Z","iopub.execute_input":"2024-03-16T19:09:05.814451Z","iopub.status.idle":"2024-03-16T19:09:24.350892Z","shell.execute_reply.started":"2024-03-16T19:09:05.814420Z","shell.execute_reply":"2024-03-16T19:09:24.350156Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import optuna\n# from sklearn.metrics import accuracy_score, roc_auc_score\n\n# def objective(trial):\n#     dtrain = xgb.DMatrix(X_train, label=y_train, enable_categorical=True)\n#     dvalid = xgb.DMatrix(X_valid, label=y_valid, enable_categorical=True)\n\n#     param = {\n#         \"verbosity\": 0,\n#         \"objective\": \"binary:logistic\",\n#         # use exact for small dataset.\n#         \"tree_method\": \"exact\",\n#         # defines booster, gblinear for linear functions.\n#         \"booster\": trial.suggest_categorical(\"booster\", [\"gbtree\", \"gblinear\", \"dart\"]),\n#         # L2 regularization weight.\n#         \"lambda\": trial.suggest_float(\"lambda\", 1e-8, 1.0, log=True),\n#         # L1 regularization weight.\n#         \"alpha\": trial.suggest_float(\"alpha\", 1e-8, 1.0, log=True),\n#         # sampling ratio for training data.\n#         \"subsample\": trial.suggest_float(\"subsample\", 0.2, 1.0),\n#         # sampling according to each tree.\n#         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.2, 1.0),\n#     }\n\n#     if param[\"booster\"] in [\"gbtree\", \"dart\"]:\n#         # maximum depth of the tree, signifies complexity of the tree.\n#         param[\"max_depth\"] = trial.suggest_int(\"max_depth\", 3, 9, step=2)\n#         # minimum child weight, larger the term more conservative the tree.\n#         param[\"min_child_weight\"] = trial.suggest_int(\"min_child_weight\", 2, 10)\n#         param[\"eta\"] = trial.suggest_float(\"eta\", 1e-8, 1.0, log=True)\n#         # defines how selective algorithm is.\n#         param[\"gamma\"] = trial.suggest_float(\"gamma\", 1e-8, 1.0, log=True)\n#         param[\"grow_policy\"] = trial.suggest_categorical(\"grow_policy\", [\"depthwise\", \"lossguide\"])\n\n#     if param[\"booster\"] == \"dart\":\n#         param[\"sample_type\"] = trial.suggest_categorical(\"sample_type\", [\"uniform\", \"weighted\"])\n#         param[\"normalize_type\"] = trial.suggest_categorical(\"normalize_type\", [\"tree\", \"forest\"])\n#         param[\"rate_drop\"] = trial.suggest_float(\"rate_drop\", 1e-8, 1.0, log=True)\n#         param[\"skip_drop\"] = trial.suggest_float(\"skip_drop\", 1e-8, 1.0, log=True)\n\n#     bst = xgb.train(param, dtrain)\n#     preds = bst.predict(dvalid)\n#     aucroc = roc_auc_score(y_valid, preds)\n#     return aucroc\n\n\n# study = optuna.create_study(direction=\"maximize\")\n# study.optimize(objective, n_trials=100, timeout=600)\n\n# print(\"Number of finished trials: \", len(study.trials))\n# print(\"Best trial:\")\n# trial = study.best_trial\n\n# print(\"  Value: {}\".format(trial.value))\n# print(\"  Params: \")\n# for key, value in trial.params.items():\n#     print(\"    {}: {}\".format(key, value))\n\n# # PLAIN OPTUNA SEARCH HAS NOT IMPROVED THE SITUATION,\n# # ACTUALLY MANUAL PARAMETERS SHOWED BETTER RESULTS FURTHER ON","metadata":{"execution":{"iopub.status.busy":"2024-03-17T14:55:10.040100Z","iopub.execute_input":"2024-03-17T14:55:10.040916Z","iopub.status.idle":"2024-03-17T14:55:10.046988Z","shell.execute_reply.started":"2024-03-17T14:55:10.040886Z","shell.execute_reply":"2024-03-17T14:55:10.045990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import xgboost as xgb\n# from sklearn.preprocessing import StandardScaler\n# from sklearn.utils import class_weight\n\n# dtrain = xgb.DMatrix(X_train, label=y_train, enable_categorical=True)\n# dvalid = xgb.DMatrix(X_valid, label=y_valid, enable_categorical=True)\n# dtest = xgb.DMatrix(X_test, label=y_test, enable_categorical=True)\n\n# xgb_params = {\n#     \"booster\": \"gbtree\",\n#     \"learning_rate\": 0.1,\n#     \"objective\": 'binary:logistic',\n#     \"n_estimators\": 1000,\n#     \"lambda\": 0.1,\n#     \"alpha\": 0.1,\n#     \"subsample\": 1.0,\n#     \"colsample_bytree\": 0.5,\n#     \"max_depth\": 7,\n#     \"min_child_weight\": 10,\n#     \"eta\": 0.1,\n#     \"gamma\": 0.5,\n#     \"grow_policy\": \"depthwise\",\n#     \"device\": 'cuda'\n# }\n\n# bst = xgb.train(xgb_params, dtrain)","metadata":{"execution":{"iopub.status.busy":"2024-03-16T18:59:41.439780Z","iopub.execute_input":"2024-03-16T18:59:41.440165Z","iopub.status.idle":"2024-03-16T18:59:43.493354Z","shell.execute_reply.started":"2024-03-16T18:59:41.440127Z","shell.execute_reply":"2024-03-16T18:59:43.492475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# graph = xgb.to_graphviz(xgbmodel, num_trees=1)\n# graph","metadata":{"execution":{"iopub.status.busy":"2024-03-16T18:59:43.504136Z","iopub.execute_input":"2024-03-16T18:59:43.504596Z","iopub.status.idle":"2024-03-16T18:59:43.509911Z","shell.execute_reply.started":"2024-03-16T18:59:43.504569Z","shell.execute_reply":"2024-03-16T18:59:43.509091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evaluation with AUC and then comparison with the stability metric is shown below.","metadata":{}},{"cell_type":"code","source":"for base, X in [(base_train, X_train), (base_valid, X_valid), (base_test, X_test)]:\n    y_pred = gbm.predict(X, num_iteration=gbm.best_iteration)\n#     y_pred = catboost_model2.predict(X)\n#     y_pred = xgbmodel.predict_proba(X)[:, 1]\n#     y_pred = model.predict(X)\n#     y_pred = model.predict_proba(X)[:, 1]\n\n#     d = xgb.DMatrix(X, enable_categorical=True)\n#     y_pred = bst.predict(d)\n    \n#     print(y_pred.sum(), y_pred.shape)\n    base[\"score\"] = y_pred\n\nprint(f'The AUC score on the train set is: {roc_auc_score(base_train[\"target\"], base_train[\"score\"])}') \nprint(f'The AUC score on the valid set is: {roc_auc_score(base_valid[\"target\"], base_valid[\"score\"])}') \nprint(f'The AUC score on the test set is: {roc_auc_score(base_test[\"target\"], base_test[\"score\"])}')  ","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:55:31.173200Z","iopub.execute_input":"2024-03-18T08:55:31.174092Z","iopub.status.idle":"2024-03-18T08:55:40.215971Z","shell.execute_reply.started":"2024-03-18T08:55:31.174045Z","shell.execute_reply":"2024-03-18T08:55:40.215007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(base, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n\nstability_score_train = gini_stability(base_train)\nstability_score_valid = gini_stability(base_valid)\nstability_score_test = gini_stability(base_test)\n\nprint(f'The stability score on the train set is: {stability_score_train}') \nprint(f'The stability score on the valid set is: {stability_score_valid}') \nprint(f'The stability score on the test set is: {stability_score_test}')","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:55:43.147827Z","iopub.execute_input":"2024-03-18T08:55:43.148514Z","iopub.status.idle":"2024-03-18T08:55:44.207799Z","shell.execute_reply.started":"2024-03-18T08:55:43.148452Z","shell.execute_reply":"2024-03-18T08:55:44.206792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:36:19.544356Z","iopub.execute_input":"2024-03-17T21:36:19.544816Z","iopub.status.idle":"2024-03-17T21:36:19.551652Z","shell.execute_reply.started":"2024-03-17T21:36:19.544759Z","shell.execute_reply":"2024-03-17T21:36:19.550629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i, (x, y) in enumerate(zip(X_train.dtypes, X_test.dtypes)):\n#     assert x == y, f'{i}, {type(x)} {X_train.columns[i]}, {X_train.dtypes[i]}, {X_test.dtypes[i]}'","metadata":{"execution":{"iopub.status.busy":"2024-03-17T19:27:36.442877Z","iopub.execute_input":"2024-03-17T19:27:36.443237Z","iopub.status.idle":"2024-03-17T19:27:36.452887Z","shell.execute_reply.started":"2024-03-17T19:27:36.443202Z","shell.execute_reply":"2024-03-17T19:27:36.452134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import sys\n\n# # These are the usual ipython objects, including this one you are creating\n# ipython_vars = ['In', 'Out', 'exit', 'quit', 'get_ipython', 'ipython_vars']\n\n# # Get a sorted list of the objects and their sizes\n# sorted([(x, sys.getsizeof(globals().get(x))) for x in dir() if not x.startswith('_') and x not in sys.modules and x not in ipython_vars], key=lambda x: x[1], reverse=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:55:51.993751Z","iopub.execute_input":"2024-03-18T08:55:51.994487Z","iopub.status.idle":"2024-03-18T08:55:51.998839Z","shell.execute_reply.started":"2024-03-18T08:55:51.994450Z","shell.execute_reply":"2024-03-18T08:55:51.997708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del df_train\n\n# del X\n# del X_train\n# del X_valid\n# del X_test\n\n# del y\n# del y_train\n# del y_valid\n# del y_test","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:36:30.134385Z","iopub.execute_input":"2024-03-17T21:36:30.135329Z","iopub.status.idle":"2024-03-17T21:36:30.145495Z","shell.execute_reply.started":"2024-03-17T21:36:30.135282Z","shell.execute_reply":"2024-03-17T21:36:30.144515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import sys\n\n# # These are the usual ipython objects, including this one you are creating\n# ipython_vars = ['In', 'Out', 'exit', 'quit', 'get_ipython', 'ipython_vars']\n\n# # Get a sorted list of the objects and their sizes\n# sorted([(x, sys.getsizeof(globals().get(x))) for x in dir() if not x.startswith('_') and x not in sys.modules and x not in ipython_vars], key=lambda x: x[1], reverse=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:56:08.856367Z","iopub.execute_input":"2024-03-18T08:56:08.857271Z","iopub.status.idle":"2024-03-18T08:56:08.861559Z","shell.execute_reply.started":"2024-03-18T08:56:08.857222Z","shell.execute_reply":"2024-03-18T08:56:08.860485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FALLS BY ML ON THE SUBMISSION TEST DATASET\n\n# df_test = feature_eng(**test_data_store)\n# df_test = df_test.select(test_columns)\n# df_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:36:43.667603Z","iopub.execute_input":"2024-03-17T21:36:43.668142Z","iopub.status.idle":"2024-03-17T21:36:43.771993Z","shell.execute_reply.started":"2024-03-17T21:36:43.668099Z","shell.execute_reply":"2024-03-17T21:36:43.771230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import sys\n\n# # These are the usual ipython objects, including this one you are creating\n# ipython_vars = ['In', 'Out', 'exit', 'quit', 'get_ipython', 'ipython_vars']\n\n# # Get a sorted list of the objects and their sizes\n# sorted([(x, sys.getsizeof(globals().get(x))) for x in dir() if not x.startswith('_') and x not in sys.modules and x not in ipython_vars], key=lambda x: x[1], reverse=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:56:14.511378Z","iopub.execute_input":"2024-03-18T08:56:14.512116Z","iopub.status.idle":"2024-03-18T08:56:14.516291Z","shell.execute_reply.started":"2024-03-18T08:56:14.512075Z","shell.execute_reply":"2024-03-18T08:56:14.515276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_test = df_test.drop(columns=[\"WEEK_NUM\"])\n# X_test = X_test.set_index(\"case_id\")\n\n# y_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:36:56.668362Z","iopub.execute_input":"2024-03-17T21:36:56.668711Z","iopub.status.idle":"2024-03-17T21:36:56.739637Z","shell.execute_reply.started":"2024-03-17T21:36:56.668687Z","shell.execute_reply":"2024-03-17T21:36:56.738858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train['max_contaddr_matchlist_1032L'], X_test['max_contaddr_matchlist_1032L']","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:36:59.582625Z","iopub.execute_input":"2024-03-17T21:36:59.583469Z","iopub.status.idle":"2024-03-17T21:36:59.587213Z","shell.execute_reply.started":"2024-03-17T21:36:59.583436Z","shell.execute_reply":"2024-03-17T21:36:59.586270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\n# df_subm = df_subm.set_index(\"case_id\")\n\n# df_subm[\"score\"] = y_pred","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:37:00.758758Z","iopub.execute_input":"2024-03-17T21:37:00.759659Z","iopub.status.idle":"2024-03-17T21:37:00.770022Z","shell.execute_reply.started":"2024-03-17T21:37:00.759624Z","shell.execute_reply":"2024-03-17T21:37:00.769116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-17T21:37:02.059404Z","iopub.execute_input":"2024-03-17T21:37:02.060121Z","iopub.status.idle":"2024-03-17T21:37:02.067263Z","shell.execute_reply.started":"2024-03-17T21:37:02.060084Z","shell.execute_reply":"2024-03-17T21:37:02.066378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !git clone https://github.com/yandex-research/tabular-dl-tabr\n# !cd /kaggle/working/tabular-dl-tabr\n# !micromamba create -f environment.yaml\n# !micromamba activate tabr","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !unzip /kaggle/input/tabr-github/tabular-dl-tabr-main","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cd /kaggle/input/tabr-github/tabular-dl-tabr-main","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !conda install environment.yaml","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission\n\nScoring the submission dataset is below, we need to take care of new categories. Then we save the score as a last step. ","metadata":{}},{"cell_type":"code","source":"X_submission = data_submission[cols_pred].to_pandas()\nX_submission = convert_strings(X_submission)\ncategorical_cols = X_train.select_dtypes(include=['category']).columns\n\n# norm_train_data.iloc[:, num_indices] = norm_train_data.iloc[:, num_indices]\n\n# num_indices = [i for i, x in enumerate(X_train.dtypes) if x == 'float64']\n\n# for i in num_indices:\n#     X_submission[c] = (X_submission[c] - X_submission[c].min()) / (X_submission[c].max() - X_submission[c].min()) + 1\n#     X_submission[c] = np.log(X_submission[c])\n\n# for i, column_name in enumerate(X_train.columns):\n#     if i in cat_indices:\n#         X_submission[column_name] = X_submission[column_name].fillna(X_submission[column_name].min(axis=0))\n#     else:\n#         X_submission[column_name] = X_submission[column_name].fillna(X_submission[column_name].min(axis=0) - 1)\n\nfor col in categorical_cols:\n    train_categories = set(X_train[col].cat.categories)\n    submission_categories = set(X_submission[col].cat.categories)\n    new_categories = submission_categories - train_categories\n    X_submission.loc[X_submission[col].isin(new_categories), col] = \"Unknown\"\n    new_dtype = pd.CategoricalDtype(categories=train_categories, ordered=True)\n    X_train[col] = X_train[col].astype(new_dtype)\n    X_submission[col] = X_submission[col].astype(new_dtype)\n\ny_submission_pred = gbm.predict(X_submission, num_iteration=gbm.best_iteration)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T19:20:00.549085Z","iopub.execute_input":"2024-03-18T19:20:00.549640Z","iopub.status.idle":"2024-03-18T19:20:00.654512Z","shell.execute_reply.started":"2024-03-18T19:20:00.549612Z","shell.execute_reply":"2024-03-18T19:20:00.653669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"case_id\": data_submission[\"case_id\"].to_numpy(),\n    \"score\": y_submission_pred\n}).set_index('case_id')\nsubmission.to_csv(\"./submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-18T19:20:00.655814Z","iopub.execute_input":"2024-03-18T19:20:00.656409Z","iopub.status.idle":"2024-03-18T19:20:00.665785Z","shell.execute_reply.started":"2024-03-18T19:20:00.656380Z","shell.execute_reply":"2024-03-18T19:20:00.664694Z"},"trusted":true},"execution_count":null,"outputs":[]}]}