{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Соревновательный анализ данных. ДЗ 2.\n\n_Всеволод Савинский_\n\nСначала я взял за основу самый залайканный ноутбук: https://www.kaggle.com/code/jetakow/home-credit-2024-starter-notebook\n\nЧто я попробал:\n* Отправил посылку без изменений, она набрала 0.36.\n* Добавил catboost вместо light gbm. Сделал специальные веса, чтобы бустинг хорошо обучался на 1, так как в датасете перевес их всего 3% от выборки.\n* Добавил оптуну, чтобы перебрать лучшие гиперпараметры для катбуста.\n* Добавил WEEK_NUM и MONTH_NUM, чтобы модель возможно нашла тренд по данным.\n* Эти улучшения только понизили качество, так как catboost очень плохо работает с дизбалансом классов.\n* Поменял catboost на light gbm, это подняло скор до 0.3\n\nДалее я взял этот ноутбук: https://www.kaggle.com/code/greysky/home-credit-baseline\nтак как там читаются больше табличек, соответственно больше данных для обучения:\n* Добавил оптуну\n* К сожалению, хоть и на локальном запуске все работает и получается довольно неплохое качество, при посылке в систему падает ошибка out of memory, хотя даже .csv файл успевает записаться. Видимо это какая-то проблема каггла. \n* В итоге после добавления оптуны качество немного поднялось.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"from pathlib import Path\nimport polars as pl\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom glob import glob \nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler\nfrom catboost import CatBoostClassifier, Pool\nimport optuna\nimport gc\nfrom tqdm.auto import tqdm\n\n\nROOT            = Path(\"/kaggle/input/home-credit-credit-risk-model-stability\")\nTRAIN_DIR       = ROOT / \"parquet_files\" / \"train\"\nTEST_DIR        = ROOT / \"parquet_files\" / \"test\"","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:18:36.280268Z","iopub.execute_input":"2024-03-17T10:18:36.280721Z","iopub.status.idle":"2024-03-17T10:18:41.215413Z","shell.execute_reply.started":"2024-03-17T10:18:36.280688Z","shell.execute_reply":"2024-03-17T10:18:41.214467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int32))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                df = df.with_columns(pl.col(col).cast(pl.Float32))\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df\n    \n    @staticmethod\n    def filter_cols(df):\n        for col in df.columns:\n            if col not in [\"target\", \"case_id\", \"WEEK_NUM\"]:\n                isnull = df[col].is_null().mean()\n\n                if isnull > 0.95:\n                    df = df.drop(col)\n\n        for col in df.columns:\n            if (col not in [\"target\", \"case_id\", \"WEEK_NUM\"]) & (df[col].dtype == pl.String):\n                freq = df[col].n_unique()\n\n                if (freq == 1) | (freq > 200):\n                    df = df.drop(col)\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:18:41.217334Z","iopub.execute_input":"2024-03-17T10:18:41.218049Z","iopub.status.idle":"2024-03-17T10:18:41.239591Z","shell.execute_reply.started":"2024-03-17T10:18:41.218006Z","shell.execute_reply":"2024-03-17T10:18:41.238178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def date_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"D\",)]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n\n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.date_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:18:41.241011Z","iopub.execute_input":"2024-03-17T10:18:41.244510Z","iopub.status.idle":"2024-03-17T10:18:41.260308Z","shell.execute_reply.started":"2024-03-17T10:18:41.244471Z","shell.execute_reply":"2024-03-17T10:18:41.259158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path, depth=None):\n    df = pl.read_parquet(path)\n    df = df.pipe(Pipeline.set_table_dtypes)\n    \n    if depth in [1, 2]:\n        df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n    \n    return df\n\n\ndef read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    return df\n\n\ndef feature_eng(df_base, depth_0, depth_1, depth_2):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday(),\n        )\n    )\n        \n    for i, df in enumerate(tqdm(depth_0 + depth_1 + depth_2)):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n        \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base\n\ndef to_pandas(df_data, cat_cols=None):\n    df_data = df_data.to_pandas()\n    \n    if cat_cols is None:\n        cat_cols = list(df_data.select_dtypes(\"object\").columns)\n    \n    df_data[cat_cols] = df_data[cat_cols].astype(\"category\")\n    \n    return df_data, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:18:41.263393Z","iopub.execute_input":"2024-03-17T10:18:41.263767Z","iopub.status.idle":"2024-03-17T10:18:41.281361Z","shell.execute_reply.started":"2024-03-17T10:18:41.263734Z","shell.execute_reply":"2024-03-17T10:18:41.280144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TRAIN_DIR / \"train_base.parquet\"),\n    \"depth_0\": [\n        read_file(TRAIN_DIR / \"train_static_cb_0.parquet\"),\n        read_files(TRAIN_DIR / \"train_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TRAIN_DIR / \"train_applprev_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_a_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_tax_registry_c_1.parquet\", 1),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_other_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_person_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_deposit_1.parquet\", 1),\n        read_file(TRAIN_DIR / \"train_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TRAIN_DIR / \"train_credit_bureau_b_2.parquet\", 2),\n        read_files(TRAIN_DIR / \"train_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}\n\ndf_train = feature_eng(**data_store)\n\nprint(\"train data shape:\\t\", df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:18:41.283486Z","iopub.execute_input":"2024-03-17T10:18:41.283848Z","iopub.status.idle":"2024-03-17T10:22:17.664808Z","shell.execute_reply.started":"2024-03-17T10:18:41.283816Z","shell.execute_reply":"2024-03-17T10:22:17.662922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_store = {\n    \"df_base\": read_file(TEST_DIR / \"test_base.parquet\"),\n    \"depth_0\": [\n        read_file(TEST_DIR / \"test_static_cb_0.parquet\"),\n        read_files(TEST_DIR / \"test_static_0_*.parquet\"),\n    ],\n    \"depth_1\": [\n        read_files(TEST_DIR / \"test_applprev_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_a_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_tax_registry_c_1.parquet\", 1),\n        read_files(TEST_DIR / \"test_credit_bureau_a_1_*.parquet\", 1),\n        read_file(TEST_DIR / \"test_credit_bureau_b_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_other_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_person_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_deposit_1.parquet\", 1),\n        read_file(TEST_DIR / \"test_debitcard_1.parquet\", 1),\n    ],\n    \"depth_2\": [\n        read_file(TEST_DIR / \"test_credit_bureau_b_2.parquet\", 2),\n        read_files(TEST_DIR / \"test_credit_bureau_a_2_*.parquet\", 2),\n    ]\n}\ndf_test = feature_eng(**data_store)\n\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:17.667957Z","iopub.execute_input":"2024-03-17T10:22:17.668548Z","iopub.status.idle":"2024-03-17T10:22:18.557662Z","shell.execute_reply.started":"2024-03-17T10:22:17.668489Z","shell.execute_reply":"2024-03-17T10:22:18.556765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_store\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:18.559015Z","iopub.execute_input":"2024-03-17T10:22:18.559370Z","iopub.status.idle":"2024-03-17T10:22:18.757954Z","shell.execute_reply.started":"2024-03-17T10:22:18.559340Z","shell.execute_reply":"2024-03-17T10:22:18.756584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.pipe(Pipeline.filter_cols)\ndf_test = df_test.select([col for col in df_train.columns if col != \"target\"])\n\nprint(\"train data shape:\\t\", df_train.shape)\nprint(\"test data shape:\\t\", df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:18.759041Z","iopub.execute_input":"2024-03-17T10:22:18.759434Z","iopub.status.idle":"2024-03-17T10:22:22.364542Z","shell.execute_reply.started":"2024-03-17T10:22:18.759402Z","shell.execute_reply":"2024-03-17T10:22:22.363390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train, cat_cols = to_pandas(df_train)\ndf_test, cat_cols = to_pandas(df_test, cat_cols)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:22.365943Z","iopub.execute_input":"2024-03-17T10:22:22.366339Z","iopub.status.idle":"2024-03-17T10:22:45.450570Z","shell.execute_reply.started":"2024-03-17T10:22:22.366304Z","shell.execute_reply":"2024-03-17T10:22:45.449215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_small, _ = train_test_split(df_train, test_size=0.9, random_state=42)\ndf_train_small, df_valid = train_test_split(df_train_small, test_size=0.02, random_state=42)\n\ndel df_train\n\ngc.collect()\n\ndf_train_small","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:45.454912Z","iopub.execute_input":"2024-03-17T10:22:45.455374Z","iopub.status.idle":"2024-03-17T10:22:54.056411Z","shell.execute_reply.started":"2024-03-17T10:22:45.455336Z","shell.execute_reply":"2024-03-17T10:22:54.054955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test\nX_test","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:54.058306Z","iopub.execute_input":"2024-03-17T10:22:54.058713Z","iopub.status.idle":"2024-03-17T10:22:54.102118Z","shell.execute_reply.started":"2024-03-17T10:22:54.058679Z","shell.execute_reply":"2024-03-17T10:22:54.101002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, y_train = df_train_small.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"]), df_train_small[\"target\"]\nX_valid, y_valid = df_valid.drop(columns=[\"target\", \"case_id\", \"WEEK_NUM\"]), df_valid[\"target\"]\nX_test = df_test.drop(columns=[\"case_id\", \"WEEK_NUM\"])\n\nprint(f\"Train: {X_train.shape}\")\nprint(f\"Valid: {X_valid.shape}\")\nprint(f\"Test: {X_test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:54.104191Z","iopub.execute_input":"2024-03-17T10:22:54.104561Z","iopub.status.idle":"2024-03-17T10:22:54.220696Z","shell.execute_reply.started":"2024-03-17T10:22:54.104529Z","shell.execute_reply":"2024-03-17T10:22:54.219194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training CatBoost\n\nUsing CatBoost to train model classifier. At first lets find good hyperparameters.","metadata":{}},{"cell_type":"code","source":"def gini_stability(X_valid, w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = X_valid.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]] \\\n        .sort_values(\"WEEK_NUM\") \\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]] \\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    \n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    \n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:54.222021Z","iopub.execute_input":"2024-03-17T10:22:54.222389Z","iopub.status.idle":"2024-03-17T10:22:54.232072Z","shell.execute_reply.started":"2024-03-17T10:22:54.222356Z","shell.execute_reply":"2024-03-17T10:22:54.230636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predefined = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"binary\",\n    \"metric\": \"auc\",\n    \"n_estimators\": 1000,\n    \"verbose\": -1,\n}","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:54.233677Z","iopub.execute_input":"2024-03-17T10:22:54.234071Z","iopub.status.idle":"2024-03-17T10:22:54.245176Z","shell.execute_reply.started":"2024-03-17T10:22:54.234037Z","shell.execute_reply":"2024-03-17T10:22:54.243548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial):\n    params = {\n        **predefined,\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.005, 0.2),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 4, 10),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.5, 1),\n        \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.5, 1),\n        \"feature_fraction_bynode\": trial.suggest_float(\"feature_fraction_bynode\", 0.5, 1),\n    }\n\n    gbm = lgb.train(\n        params,\n        lgb_train,\n        valid_sets=lgb_valid,\n        callbacks=[lgb.log_evaluation(50), lgb.early_stopping(10)]\n    )\n    \n    model = lgb.LGBMClassifier(**params)\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        callbacks=[lgb.log_evaluation(100), lgb.early_stopping(25)]\n    )\n    \n    y_pred = model.predict(X_valid)\n    return roc_auc_score(y_pred, y)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:54.247048Z","iopub.execute_input":"2024-03-17T10:22:54.247610Z","iopub.status.idle":"2024-03-17T10:22:54.259801Z","shell.execute_reply.started":"2024-03-17T10:22:54.247555Z","shell.execute_reply":"2024-03-17T10:22:54.258461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# study = optuna.create_study(direction=\"maximize\")\n# study.optimize(objective, n_trials=10, timeout=3000)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:54.261478Z","iopub.execute_input":"2024-03-17T10:22:54.261939Z","iopub.status.idle":"2024-03-17T10:22:54.271658Z","shell.execute_reply.started":"2024-03-17T10:22:54.261897Z","shell.execute_reply":"2024-03-17T10:22:54.270566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# params = study.best_trial.params\nparams = {'learning_rate': 0.10962189150972917, 'max_depth': 7, 'feature_fraction': 0.6258408744117396, 'bagging_fraction': 0.8292966834406343}\nprint(params)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:54.273553Z","iopub.execute_input":"2024-03-17T10:22:54.274261Z","iopub.status.idle":"2024-03-17T10:22:54.291981Z","shell.execute_reply.started":"2024-03-17T10:22:54.273993Z","shell.execute_reply":"2024-03-17T10:22:54.290773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    **predefined,\n    **params,\n}\n\nmodel = lgb.LGBMClassifier(**params)\nmodel.fit(\n    X_train, y_train,\n    eval_set=[(X_valid, y_valid)],\n    callbacks=[lgb.log_evaluation(100), lgb.early_stopping(25)]\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:22:54.293585Z","iopub.execute_input":"2024-03-17T10:22:54.293910Z","iopub.status.idle":"2024-03-17T10:23:21.922109Z","shell.execute_reply.started":"2024-03-17T10:22:54.293882Z","shell.execute_reply":"2024-03-17T10:23:21.920927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(X_train)\nprint(pred.mean(), y_train.mean())","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:23:21.923629Z","iopub.execute_input":"2024-03-17T10:23:21.924001Z","iopub.status.idle":"2024-03-17T10:23:31.264695Z","shell.execute_reply.started":"2024-03-17T10:23:21.923973Z","shell.execute_reply":"2024-03-17T10:23:31.263505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evaluation with AUC and then comparison with the stability metric is shown below.","metadata":{}},{"cell_type":"code","source":"for name, X, y in [(\"train\", X_train, y_train), (\"valid\", X_valid, y_valid)]:\n    y_pred = model.predict(X)\n    print(f'The AUC score on the {name} set is: {roc_auc_score(y_pred, y)}') ","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:23:31.266772Z","iopub.execute_input":"2024-03-17T10:23:31.267637Z","iopub.status.idle":"2024-03-17T10:23:40.744580Z","shell.execute_reply.started":"2024-03-17T10:23:31.267592Z","shell.execute_reply":"2024-03-17T10:23:40.743409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission\n\nScoring the submission dataset is below, we need to take care of new categories. Then we save the score as a last step. ","metadata":{}},{"cell_type":"code","source":"X_test = df_test.drop(columns=[\"WEEK_NUM\"])\nX_test = X_test.set_index(\"case_id\")\n\ny_pred = pd.Series(model.predict_proba(X_test)[:, 1], index=X_test.index)\n\ndf_subm = pd.read_csv(ROOT / \"sample_submission.csv\")\ndf_subm = df_subm.set_index(\"case_id\")\n\ndf_subm[\"score\"] = y_pred\n\ndf_subm","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:27:27.400325Z","iopub.execute_input":"2024-03-17T10:27:27.400822Z","iopub.status.idle":"2024-03-17T10:27:27.534772Z","shell.execute_reply.started":"2024-03-17T10:27:27.400787Z","shell.execute_reply":"2024-03-17T10:27:27.533843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check null: \", df_subm[\"score\"].isnull().any())\n\ndf_subm.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-17T10:23:40.853911Z","iopub.execute_input":"2024-03-17T10:23:40.854448Z","iopub.status.idle":"2024-03-17T10:23:40.868014Z","shell.execute_reply.started":"2024-03-17T10:23:40.854401Z","shell.execute_reply":"2024-03-17T10:23:40.866525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Submission has been completed!","metadata":{}}]}