{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":8066561,"sourceType":"datasetVersion","datasetId":4660255},{"sourceId":161936385,"sourceType":"kernelVersion"}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n!pip install --force-reinstall scikit-learn --no-index --find-links=file:///kaggle/input/scikit-learn-1-4-0/ ","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:32:57.295381Z","iopub.execute_input":"2024-04-12T13:32:57.295802Z","iopub.status.idle":"2024-04-12T13:33:34.839739Z","shell.execute_reply.started":"2024-04-12T13:32:57.295749Z","shell.execute_reply":"2024-04-12T13:33:34.838474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\nfrom glob import glob\nimport os\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom lightgbm import LGBMClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import SVC\nfrom sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nimport optuna\nimport functools\nfrom optuna.logging import get_logger\nfrom catboost import CatBoostClassifier\nfrom fastai.tabular.all import *\nimport warnings\nfrom time import perf_counter\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.model_selection import cross_validate\nimport gc\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:33:34.841934Z","iopub.execute_input":"2024-04-12T13:33:34.842255Z","iopub.status.idle":"2024-04-12T13:33:44.165057Z","shell.execute_reply.started":"2024-04-12T13:33:34.842229Z","shell.execute_reply":"2024-04-12T13:33:44.164095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn\nassert sklearn.__version__ == '1.4.0'","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:33:44.166173Z","iopub.execute_input":"2024-04-12T13:33:44.166714Z","iopub.status.idle":"2024-04-12T13:33:44.171329Z","shell.execute_reply.started":"2024-04-12T13:33:44.166688Z","shell.execute_reply":"2024-04-12T13:33:44.170376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/working/models\"\nTRIAL_BASE_DIR = \"trials\"\nCAT_DIR = BASE_DIR + \"/\" + \"catboost\"\nLGBM_DIR = BASE_DIR + \"/\" + \"lightgbm\"\n# fastai prepends models to the path\nNN_DIR = BASE_DIR + \"/\" + \"nn\"","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:33:44.182926Z","iopub.execute_input":"2024-04-12T13:33:44.183206Z","iopub.status.idle":"2024-04-12T13:33:44.189140Z","shell.execute_reply.started":"2024-04-12T13:33:44.183177Z","shell.execute_reply":"2024-04-12T13:33:44.188125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir(f\"{BASE_DIR}\")\nos.mkdir(f\"{TRIAL_BASE_DIR}\")\nos.mkdir(f\"{CAT_DIR}\")\nos.mkdir(f\"{LGBM_DIR}\")\n# fastai prepends models to the path\nos.mkdir(f\"{NN_DIR}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:33:44.190287Z","iopub.execute_input":"2024-04-12T13:33:44.192921Z","iopub.status.idle":"2024-04-12T13:33:44.198462Z","shell.execute_reply.started":"2024-04-12T13:33:44.192884Z","shell.execute_reply":"2024-04-12T13:33:44.197666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRIAL_BASE_DIR = \"sqlite:////kaggle/working\" + \"/\" + \"trials\"\n\nNN_STORAGE_URL = f\"{TRIAL_BASE_DIR}\" + \"/\" + \"nn_trial.db\"\nLGBM_STORAGE_URL = f\"{TRIAL_BASE_DIR}\" + \"/\" + \"lgbm_trial.db\"\nCAT_STORAGE_URL = f\"{TRIAL_BASE_DIR}\" + \"/\" + \"cat_trial.db\"\nRF_STORAGE_URL = f\"{TRIAL_BASE_DIR}\" + \"/\" + \"rf_trial.db\"\nSVC_STORAGE_URL = f\"{TRIAL_BASE_DIR}\" + \"/\" + \"svc_trial.db\"\nSTACKING_STORAGE_URL = f\"{TRIAL_BASE_DIR}\" + \"/\" + \"stacking_trial.db\"","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:33:44.199544Z","iopub.execute_input":"2024-04-12T13:33:44.201162Z","iopub.status.idle":"2024-04-12T13:33:44.206566Z","shell.execute_reply.started":"2024-04-12T13:33:44.201137Z","shell.execute_reply":"2024-04-12T13:33:44.205808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False\nOPTIMIZE = False\nOPTIMIZE_STACKING = False\nINFER = True\nFEATURE_ENG = False\nVALIDATE_ENSEMBLE = False\nTRAIN_AND_SAVE_BEST_CONFIG = False\nLOAD_AND_PREDICT = False\nTRAIN_AND_SAVE_BEST_CONFIG_FOR_ONE_MODEL = True\nINFER_ENSEMBLE = False\nTRIALS = 10\nFOLDS = 5\nSEED=42","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:33:44.207623Z","iopub.execute_input":"2024-04-12T13:33:44.208200Z","iopub.status.idle":"2024-04-12T13:33:44.216453Z","shell.execute_reply.started":"2024-04-12T13:33:44.208171Z","shell.execute_reply":"2024-04-12T13:33:44.215652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed: int):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed_everything(SEED)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/homecredit-models/* /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:33:44.228403Z","iopub.execute_input":"2024-04-12T13:33:44.228752Z","iopub.status.idle":"2024-04-12T13:33:48.198220Z","shell.execute_reply.started":"2024-04-12T13:33:44.228720Z","shell.execute_reply":"2024-04-12T13:33:48.196469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/home-credit-credit-risk-model-stability/parquet_files\"\nTRAIN_DIR = f\"{BASE_DIR}\"+ \"/\" + \"train/\"\nTEST_DIR = f\"{BASE_DIR}\"+ \"/\" + \"test/\"","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:30:13.559228Z","iopub.execute_input":"2024-04-12T13:30:13.559596Z","iopub.status.idle":"2024-04-12T13:30:13.565139Z","shell.execute_reply.started":"2024-04-12T13:30:13.559564Z","shell.execute_reply":"2024-04-12T13:30:13.564031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from optuna.logging import disable_default_handler\ncurrent_train_model = \"EMPTY\"\n# Disable default handler of Optuna's logger\ndisable_default_handler()\n\ndef logging_callback(study, trial):\n    global current_train_model\n    current_trial = trial.number\n    if study.best_trial.number == trial.number:\n        logger = get_logger(\"optuna\")\n        logger.warning(f\"New best value: {trial.value}, Params: {trial.params}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-12T13:30:13.566623Z","iopub.execute_input":"2024-04-12T13:30:13.567046Z","iopub.status.idle":"2024-04-12T13:30:13.576049Z","shell.execute_reply.started":"2024-04-12T13:30:13.566988Z","shell.execute_reply":"2024-04-12T13:30:13.575088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Pipeline:\n    @staticmethod\n    def set_table_dtypes(df):\n        for col in df.columns:\n            if col in [\"case_id\", \"WEEK_NUM\", \"num_group1\", \"num_group2\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Int64))\n            elif col in [\"date_decision\"]:\n                df = df.with_columns(pl.col(col).cast(pl.Date))\n            elif col[-1] in (\"P\", \"A\"):\n                df = df.with_columns(pl.col(col).cast(pl.Float64))\n            elif col[-1] in (\"M\",):\n                df = df.with_columns(pl.col(col).cast(pl.String))\n            elif col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col).cast(pl.Date))            \n\n        return df\n    \n    @staticmethod\n    def handle_dates(df):\n        for col in df.columns:\n            if col[-1] in (\"D\",):\n                df = df.with_columns(pl.col(col) - pl.col(\"date_decision\"))\n                df = df.with_columns(pl.col(col).dt.total_days())\n                \n        df = df.drop(\"date_decision\", \"MONTH\")\n\n        return df","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.450503Z","iopub.status.idle":"2024-04-08T18:40:12.450856Z","shell.execute_reply.started":"2024-04-08T18:40:12.450687Z","shell.execute_reply":"2024-04-08T18:40:12.450701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Aggregator:\n    @staticmethod\n    def num_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"P\", \"A\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n        \n        return expr_max\n    \n    @staticmethod\n    def str_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"M\",)]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def other_expr(df):\n        cols = [col for col in df.columns if col[-1] in (\"T\", \"L\")]\n        \n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def count_expr(df):\n        cols = [col for col in df.columns if \"num_group\" in col]\n\n        expr_max = [pl.max(col).alias(f\"max_{col}\") for col in cols]\n\n        return expr_max\n    \n    @staticmethod\n    def get_exprs(df):\n        exprs = Aggregator.num_expr(df) + \\\n                Aggregator.str_expr(df) + \\\n                Aggregator.other_expr(df) + \\\n                Aggregator.count_expr(df)\n        return exprs","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.452137Z","iopub.status.idle":"2024-04-08T18:40:12.452479Z","shell.execute_reply.started":"2024-04-08T18:40:12.452286Z","shell.execute_reply":"2024-04-08T18:40:12.452323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_files(regex_path, depth=None):\n    chunks = []\n    for path in glob.glob(str(regex_path)):\n        df = pl.read_parquet(path)\n        df = df.pipe(Pipeline.set_table_dtypes)\n        \n        if depth in [1, 2]:\n            df = df.group_by(\"case_id\").agg(Aggregator.get_exprs(df))\n        \n        chunks.append(df)\n        \n    df = pl.concat(chunks, how=\"vertical_relaxed\")\n    df = df.unique(subset=[\"case_id\"])\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.454861Z","iopub.status.idle":"2024-04-08T18:40:12.455343Z","shell.execute_reply.started":"2024-04-08T18:40:12.455087Z","shell.execute_reply":"2024-04-08T18:40:12.455106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_eng(df_base, depth_0, depth_1):\n    df_base = (\n        df_base\n        .with_columns(\n            month_decision = pl.col(\"date_decision\").dt.month(),\n            weekday_decision = pl.col(\"date_decision\").dt.weekday()\n        )\n    )\n    \n    for i, df in enumerate(depth_0 + depth_1):\n        df_base = df_base.join(df, how=\"left\", on=\"case_id\", suffix=f\"_{i}\")\n    \n    df_base = df_base.pipe(Pipeline.handle_dates)\n    \n    return df_base","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.456338Z","iopub.status.idle":"2024-04-08T18:40:12.456758Z","shell.execute_reply.started":"2024-04-08T18:40:12.456542Z","shell.execute_reply":"2024-04-08T18:40:12.456560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_id_column(df_as_polar):\n    idx_range = [i for i in range(len(df_as_polar))]\n    return df_as_polar.with_columns(pl.Series(name=\"index\", values=idx_range))","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.458903Z","iopub.status.idle":"2024-04-08T18:40:12.459293Z","shell.execute_reply.started":"2024-04-08T18:40:12.459096Z","shell.execute_reply":"2024-04-08T18:40:12.459123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_data_store(DIR_PREFIX = TRAIN_DIR, FILE_PREFIX = \"train\"):\n    return {\n        \"df_base\": read_files(DIR_PREFIX + f\"{FILE_PREFIX}_base.parquet\"),\n        \"depth_0\": [\n            read_files(DIR_PREFIX + f\"{FILE_PREFIX}_static_cb_0.parquet\")\n        ],\n        \"depth_1\": [\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_applprev_1_*.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_tax_registry_a_1.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_tax_registry_b_1.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_tax_registry_c_1.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_credit_bureau_a_1_*.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_credit_bureau_b_1.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_other_1.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_person_1.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_deposit_1.parquet\", 1),\n          read_files(DIR_PREFIX + f\"{FILE_PREFIX}_debitcard_1.parquet\", 1),\n      ]\n    }","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.461951Z","iopub.status.idle":"2024-04-08T18:40:12.462332Z","shell.execute_reply.started":"2024-04-08T18:40:12.462143Z","shell.execute_reply":"2024-04-08T18:40:12.462157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_boolean_columns_with_int(inp):\n    for c in inp.select(pl.col(pl.Boolean)).columns:\n        c_as_int = pl.Series(name=c,values=LabelEncoder().fit_transform(inp.select(pl.col(c)).to_numpy().ravel()))\n        inp.drop(c)\n        inp = inp.with_columns(c_as_int)\n    return inp","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.463649Z","iopub.status.idle":"2024-04-08T18:40:12.463968Z","shell.execute_reply.started":"2024-04-08T18:40:12.463812Z","shell.execute_reply":"2024-04-08T18:40:12.463826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_string_columns_with_int(inp):\n    for c in inp.select(pl.col(pl.Utf8)).columns:\n        c_as_int = pl.Series(name=c,values=LabelEncoder().fit_transform(inp.select(pl.col(c)).to_numpy().ravel()))\n        inp.drop(c)\n        inp = inp.with_columns(c_as_int)\n    return inp","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.465056Z","iopub.status.idle":"2024-04-08T18:40:12.465427Z","shell.execute_reply.started":"2024-04-08T18:40:12.465216Z","shell.execute_reply":"2024-04-08T18:40:12.465229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pre_process_frame(DIR_PREFIX = TRAIN_DIR, FILE_PREFIX = \"train\"):\n    data_store = create_data_store(DIR_PREFIX, FILE_PREFIX)\n    df_train = feature_eng(**data_store)\n    df_train = replace_string_columns_with_int(df_train)\n    df_train = add_id_column(df_train)\n    return replace_boolean_columns_with_int(df_train)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.466619Z","iopub.status.idle":"2024-04-08T18:40:12.466953Z","shell.execute_reply.started":"2024-04-08T18:40:12.466786Z","shell.execute_reply":"2024-04-08T18:40:12.466800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if FEATURE_ENG:\n    df_train = pre_process_frame()\n    df_train.write_parquet(\"inp_preprocessed.parquet\")\nelse:\n    df_train = pl.read_parquet(\"inp_preprocessed.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.468411Z","iopub.status.idle":"2024-04-08T18:40:12.468762Z","shell.execute_reply.started":"2024-04-08T18:40:12.468588Z","shell.execute_reply":"2024-04-08T18:40:12.468603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gini_stability(X_test, w_fallingrate=88.0, w_resstd=-0.5):\n    if not isinstance(X_test, pd.DataFrame):\n        base = convert_to_pandas(X_test)\n    else:\n        base = X_test\n    interm = base\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", \"score\"]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", \"score\"]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[\"score\"])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.469862Z","iopub.status.idle":"2024-04-08T18:40:12.470194Z","shell.execute_reply.started":"2024-04-08T18:40:12.470020Z","shell.execute_reply":"2024-04-08T18:40:12.470033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_inp = df_train.select(pl.col(\"target\"))\nX_inp = df_train\nweeks = df_train.select(pl.col(\"WEEK_NUM\"))","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.471242Z","iopub.status.idle":"2024-04-08T18:40:12.471593Z","shell.execute_reply.started":"2024-04-08T18:40:12.471441Z","shell.execute_reply":"2024-04-08T18:40:12.471454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cont_cat_variables(df):\n    df_nn_final = df.copy()\n\n    return cont_cat_split(df_nn_final, max_card=9000, dep_var=\"target\")\n\ndef convert_to_pandas(df):\n    df_as_pandas = df.to_pandas()\n    return df_as_pandas\n\ndef prepare_nn_dataset(df):\n    df_as_pandas = df.to_pandas()\n    df_as_pandas.fillna(1e-6, inplace=True)\n    cont_nn, cat_nn = get_cont_cat_variables(df_as_pandas)\n    procs_nn = [Categorify, FillMissing, Normalize]\n    splitter = RandomSplitter(valid_pct=0.2, seed=42)\n    splits = splitter(range_of(df_as_pandas))\n    return TabularPandas(df_as_pandas, procs_nn, cat_nn, cont_nn, y_names=\"target\", splits=splits)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.472563Z","iopub.status.idle":"2024-04-08T18:40:12.472855Z","shell.execute_reply.started":"2024-04-08T18:40:12.472708Z","shell.execute_reply":"2024-04-08T18:40:12.472720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_roc_auc_score(y_score, y_true):\n    predicted_score = np.argmax(y_score.sigmoid().cpu().numpy(), axis=1)\n    try:\n        return roc_auc_score(y_true.cpu().numpy().ravel(), predicted_score)\n    except ValueError:\n        print(\"could not calculate roc_auc_score returning 0\")\n        return 0","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.474238Z","iopub.status.idle":"2024-04-08T18:40:12.474608Z","shell.execute_reply.started":"2024-04-08T18:40:12.474452Z","shell.execute_reply":"2024-04-08T18:40:12.474466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, ClassifierMixin\nfrom tqdm import tqdm\n\nclass FastAiTabularNN(BaseEstimator, ClassifierMixin):\n    def __init__(self, first_layer_size=500, second_layer_size=200, learner = None):\n        self.first_layer_size = first_layer_size\n        self.second_layer_size = second_layer_size\n        self.learn = learner\n\n    def fit(self, X, y):\n        df_as_pl = pl.concat([pl.DataFrame(X), pl.DataFrame({\"target\": y})], how=\"horizontal\")\n        to_nn = prepare_nn_dataset(df_as_pl)\n        self.dls = to_nn.dataloaders(1024)\n        self.learn = tabular_learner(self.dls, layers=[self.first_layer_size,self.second_layer_size], \n                        n_out=2, # Adjust this based on your number of categories\n                        loss_func=CrossEntropyLossFlat(), # Use CrossEntropyLossFlat for classification\n                        metrics=custom_roc_auc_score) # Consider adding metrics like accuracy for evaluation\n\n    \n        \n        with self.learn.no_bar(), self.learn.no_logging():\n            optimal_lr = self.learn.lr_find(show_plot=False)\n            optimal_lr = optimal_lr.valley\n            self.learn.fit_one_cycle(10, optimal_lr, cbs=EarlyStoppingCallback(monitor='valid_loss', min_delta=0.01, patience=3))\n        return self\n\n    def predict(self, X):\n        if not isinstance(X, pd.DataFrame):\n            X = convert_to_pandas(X)\n        test_dl = self.learn.dls.test_dl(X.fillna(1e-6))\n        _, _, decoded = self.learn.get_preds(dl=test_dl, with_decoded=True)\n        \n        return decoded.cpu().numpy()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.475782Z","iopub.status.idle":"2024-04-08T18:40:12.476084Z","shell.execute_reply.started":"2024-04-08T18:40:12.475935Z","shell.execute_reply":"2024-04-08T18:40:12.475947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_svc_classifier(**cfg):\n    \n    preprocessing_steps = [\n        SimpleImputer(fill_value=1, strategy='constant')\n        ,StandardScaler()\n    ]\n\n    pipeline_steps = preprocessing_steps + [\n        SVC(**cfg)\n    ]\n    \n    print(pipeline_steps)\n    # Create the pipeline\n    return functools.partial(make_pipeline, *pipeline_steps)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.477218Z","iopub.status.idle":"2024-04-08T18:40:12.477556Z","shell.execute_reply.started":"2024-04-08T18:40:12.477400Z","shell.execute_reply":"2024-04-08T18:40:12.477414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def input_prepare(df_inp, range_iter = 10):\n    X = df_inp.filter(pl.col(\"index\").is_in(range(range_iter)))\n    y = X.select(pl.col(\"target\"))\n    X = X.drop(\"target\")\n    return X, y","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.479045Z","iopub.status.idle":"2024-04-08T18:40:12.479422Z","shell.execute_reply.started":"2024-04-08T18:40:12.479212Z","shell.execute_reply":"2024-04-08T18:40:12.479226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_nn_configuration(trial):\n    params = {}\n    params[\"first_layer_size\"] = trial.suggest_int(\"first_layer_size\",300, 600 )\n    params[\"second_layer_size\"] = trial.suggest_int(\"second_layer_size\", 100, 300)\n    \n    return get_nn_classifier(**params)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.480896Z","iopub.status.idle":"2024-04-08T18:40:12.481194Z","shell.execute_reply.started":"2024-04-08T18:40:12.481047Z","shell.execute_reply":"2024-04-08T18:40:12.481059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_nn_classifier(**cfg):\n    return functools.partial(FastAiTabularNN, **cfg)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.482287Z","iopub.status.idle":"2024-04-08T18:40:12.482667Z","shell.execute_reply.started":"2024-04-08T18:40:12.482494Z","shell.execute_reply":"2024-04-08T18:40:12.482508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_nn_classifier(model, fold_count):\n    model.learn.export(f\"{NN_DIR}/nn_{fold_count}.bin\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.483939Z","iopub.status.idle":"2024-04-08T18:40:12.484254Z","shell.execute_reply.started":"2024-04-08T18:40:12.484099Z","shell.execute_reply":"2024-04-08T18:40:12.484112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_nn_classifier(fold):\n    learner = load_learner(NN_DIR + \"/\" + f\"nn_{fold}.bin\", cpu=False)\n    fastAiTabularNN = FastAiTabularNN(learner = learner)\n    return fastAiTabularNN","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.485247Z","iopub.status.idle":"2024-04-08T18:40:12.485620Z","shell.execute_reply.started":"2024-04-08T18:40:12.485452Z","shell.execute_reply":"2024-04-08T18:40:12.485478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_catboost_classifier(**cfg):\n    return functools.partial(CatBoostClassifier, loss_function='Logloss',task_type=\"GPU\", **cfg)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.486806Z","iopub.status.idle":"2024-04-08T18:40:12.487124Z","shell.execute_reply.started":"2024-04-08T18:40:12.486969Z","shell.execute_reply":"2024-04-08T18:40:12.486983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_catboost_classifier(model, fold_counter):\n    model.save_model(f\"{CAT_DIR}/cat_{fold_counter}.bin\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.488544Z","iopub.status.idle":"2024-04-08T18:40:12.488898Z","shell.execute_reply.started":"2024-04-08T18:40:12.488719Z","shell.execute_reply":"2024-04-08T18:40:12.488733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_catboost_classifier(fold):\n    path = f\"{CAT_DIR}/cat_{fold}.bin\"\n    loaded_model = CatBoostClassifier()\n\n    # Load the model from the file\n    \n    loaded_model.load_model(path)\n    \n    return loaded_model","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.490357Z","iopub.status.idle":"2024-04-08T18:40:12.490720Z","shell.execute_reply.started":"2024-04-08T18:40:12.490547Z","shell.execute_reply":"2024-04-08T18:40:12.490562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lgbm_classifier(**cfg):\n    return functools.partial(LGBMClassifier, boosting_type = \"gbdt\", objective='binary',metric=\"auc\", device=\"gpu\", verbose=1, random_state=SEED, **cfg)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.491694Z","iopub.status.idle":"2024-04-08T18:40:12.492010Z","shell.execute_reply.started":"2024-04-08T18:40:12.491853Z","shell.execute_reply":"2024-04-08T18:40:12.491866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_lgbm_classifier(model, fold_counter):\n    model._Booster.save_model(f\"{LGBM_DIR}/lgbm_{fold_counter}.txt\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.493173Z","iopub.status.idle":"2024-04-08T18:40:12.493655Z","shell.execute_reply.started":"2024-04-08T18:40:12.493427Z","shell.execute_reply":"2024-04-08T18:40:12.493446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_lgbm_classifier(fold):\n    path = LGBM_DIR + \"/\" + f\"lgbm_{fold}.txt\"\n    return lgb.Booster(model_file=path);","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.495190Z","iopub.status.idle":"2024-04-08T18:40:12.495601Z","shell.execute_reply.started":"2024-04-08T18:40:12.495409Z","shell.execute_reply":"2024-04-08T18:40:12.495427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_catboost_configuration(trial):\n    catboost_model_config = {\n        'iterations': trial.suggest_int('iterations', 100, 1000),\n        'depth': trial.suggest_int('depth', 4, 10),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),\n        'random_strength': trial.suggest_int('random_strength', 1, 20),\n        'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n        'border_count': trial.suggest_int('border_count', 1, 255),\n        'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 3, 8),\n        'scale_pos_weight': trial.suggest_float('scale_pos_weight', 0.01, 1.0),\n    }\n    \n    return get_catboost_classifier(**catboost_model_config)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.496703Z","iopub.status.idle":"2024-04-08T18:40:12.497027Z","shell.execute_reply.started":"2024-04-08T18:40:12.496871Z","shell.execute_reply":"2024-04-08T18:40:12.496885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lgbm_configuration(trial):\n    lgbm_model_config = {\n        \"max_depth\": trial.suggest_int(\"max_depth\", 6, 10),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 24, 40),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 5, 15),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.1, log=True),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.7, 0.9),\n        \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.7, 0.9),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 1, 10),\n        \"n_estimators\": trial.suggest_int(\"n_estimators\", 50, 150),\n        \"min_data_in_bin\": trial.suggest_int(\"min_data_in_bin\", 1, 5),\n        \"max_bin\": trial.suggest_int(\"max_bin\", 50, 100)\n    }\n\n    return get_lgbm_classifier(**lgbm_model_config)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.498346Z","iopub.status.idle":"2024-04-08T18:40:12.498685Z","shell.execute_reply.started":"2024-04-08T18:40:12.498515Z","shell.execute_reply":"2024-04-08T18:40:12.498530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_svc_configuration(trial):\n    svc_model_configuration = {\n        \"C\": trial.suggest_loguniform('C', 1e-10, 1e10),\n        \"kernel\": trial.suggest_categorical('kernel', ['linear', 'rbf', 'poly', 'sigmoid'])\n    }\n    \n    return get_svc_classifier(**svc_model_configuration)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.500739Z","iopub.status.idle":"2024-04-08T18:40:12.501188Z","shell.execute_reply.started":"2024-04-08T18:40:12.500960Z","shell.execute_reply":"2024-04-08T18:40:12.500981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_with_config(model_key):\n    storage = model_configs[model_key][\"storage\"]\n    name = model_configs[model_key][\"name\"]\n    study = optuna.create_study(study_name=name, storage=storage, load_if_exists=True, direction='maximize')\n    \n    model = model_configs[model_key][\"model_constructor\"]()\n    return model(**study.best_params)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.502612Z","iopub.status.idle":"2024-04-08T18:40:12.503057Z","shell.execute_reply.started":"2024-04-08T18:40:12.502828Z","shell.execute_reply":"2024-04-08T18:40:12.502846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_stacking_classifier(trial, base_estimators=[\"LGBM\", \"NN\"], get_final_estimator=get_catboost_configuration):\n    final_estimator=\"Cat\"\n    stacking_clf = StackingClassifier(\n        estimators=[(b, get_model_with_config(b))for b in base_estimators],\n        final_estimator=get_final_estimator(trial)(),\n        cv=FOLDS  # number of cross-validation folds\n    )\n        \n    return functools.partial(StackingClassifier, estimators=[(b, get_model_with_config(b))for b in base_estimators], final_estimator=get_final_estimator(trial)(), cv=FOLDS)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.504843Z","iopub.status.idle":"2024-04-08T18:40:12.505330Z","shell.execute_reply.started":"2024-04-08T18:40:12.505054Z","shell.execute_reply":"2024-04-08T18:40:12.505073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_stacking_configuration(trial):\n    return get_stacking_classifier(trial)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.506992Z","iopub.status.idle":"2024-04-08T18:40:12.507437Z","shell.execute_reply.started":"2024-04-08T18:40:12.507197Z","shell.execute_reply":"2024-04-08T18:40:12.507214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SuppressOutput:\n    def __enter__(self):\n        # Save the current stdout and stderr\n        self._original_stdout = sys.stdout\n        self._original_stderr = sys.stderr\n        # Open os.devnull for writing and redirect stdout and stderr there\n        self._devnull = open(os.devnull, 'w')\n        sys.stdout = self._devnull\n        sys.stderr = self._devnull\n\n    def __exit__(self, exc_type, exc_val, exc_tb):\n        # Restore the original stdout and stderr\n        sys.stdout = self._original_stdout\n        sys.stderr = self._original_stderr\n        # Close the devnull file\n        self._devnull.close()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.508615Z","iopub.status.idle":"2024-04-08T18:40:12.509127Z","shell.execute_reply.started":"2024-04-08T18:40:12.508875Z","shell.execute_reply":"2024-04-08T18:40:12.508896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_configs = {\n\"SVC\": {\n    \"name\": \"svc_study\",\n    \"storage\": SVC_STORAGE_URL,\n    \"optuna_config\": get_svc_configuration,\n    \"model_constructor\": get_svc_classifier,\n},\n\"NN\": {\n    \"name\": \"nn_study\",\n    \"storage\": NN_STORAGE_URL,\n    \"optuna_config\": get_nn_configuration,\n    \"model_constructor\": get_nn_classifier,\n    \"save_method\": save_nn_classifier,\n    \"load_method\": load_nn_classifier\n},\n\"LGBM\": {\n    \"name\": \"lgbm_study\",\n    \"storage\": LGBM_STORAGE_URL,\n    \"optuna_config\": get_lgbm_configuration,\n    \"model_constructor\": get_lgbm_classifier,\n    \"save_method\": save_lgbm_classifier,\n    \"load_method\": load_lgbm_classifier\n},\n\"Cat\": {\n    \"name\": \"cat_study\",\n    \"storage\": CAT_STORAGE_URL,\n    \"optuna_config\": get_catboost_configuration,\n    \"model_constructor\": get_catboost_classifier,\n    \"save_method\": save_catboost_classifier,\n    \"load_method\": load_catboost_classifier\n},    \n\"Stacking\": {\n    \"name\": \"stacking_study\",\n    \"storage\": STACKING_STORAGE_URL,\n    \"optuna_config\": get_stacking_configuration,\n    \"model_constructor\": get_stacking_classifier\n}\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.510501Z","iopub.status.idle":"2024-04-08T18:40:12.510880Z","shell.execute_reply.started":"2024-04-08T18:40:12.510705Z","shell.execute_reply":"2024-04-08T18:40:12.510719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare(inp, indices=None):\n    if indices is not None:\n        inp_filtered = inp.filter(pl.col(\"index\").is_in(indices))\n    else:\n        inp_filtered = inp\n    y_filtered = inp_filtered.select(pl.col(\"target\"))\n    inp_filtered = inp_filtered.drop(\"target\")\n    dropeed_week_num = inp_filtered.select(\"WEEK_NUM\")\n    inp_filtered = inp_filtered.drop(\"WEEK_NUM\")\n    \n    return inp_filtered.to_pandas(), y_filtered.to_numpy().T.ravel(), dropeed_week_num","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.512258Z","iopub.status.idle":"2024-04-08T18:40:12.512606Z","shell.execute_reply.started":"2024-04-08T18:40:12.512450Z","shell.execute_reply":"2024-04-08T18:40:12.512464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def infer(X, y, model, weeks_ = None):\n    if weeks_ is None:\n        weeks_ = weeks\n    if DEBUG:\n        print(f\"model is {model}\")\n    all_gini_scores = []\n    all_roc_scores = []\n    all_fitted_models = []\n    \n    cv = StratifiedGroupKFold(n_splits=FOLDS, shuffle=False,)\n    if DEBUG:\n        print(f\"X shape is {X.shape}\")\n        print(f\"y shape is {y.shape}\")\n    current_model = model()\n    X = X.drop(\"index\")\n    X_inp = X.drop(\"target\")\n    X_inp = X_inp.drop(\"WEEK_NUM\")\n    assert 'WEEK_NUM' in X.columns\n    X_inp_as_pd = X_inp.to_pandas()\n    y_as_pd = y.to_pandas()\n    cv_results = cross_validate(\n        model(),\n        X_inp_as_pd, y_as_pd, \n        groups=X['WEEK_NUM'], \n        scoring='roc_auc', \n        cv=cv,\n        verbose=3, \n        return_estimator=True, \n        return_indices=True\n    )\n    \n    stability_results = []\n    X = X.to_pandas()\n    for fold, (idx, model) in enumerate(zip(cv_results['indices']['test'], cv_results['estimator'])):\n        df_res = pd.DataFrame()\n        #current_train = prepare(X, idx)\n        df_res['WEEK_NUM'] = X['WEEK_NUM'].iloc[idx].values\n        df_res['target'] = X['target'].iloc[idx].values\n        inp = X.iloc[idx]\n        del inp[\"target\"]\n        del inp[\"WEEK_NUM\"]\n        df_res['score'] = model.predict_proba(inp)[:, 1]\n        df_res['fold'] = fold\n        stability_results.append(df_res)\n    \n    #for idx, (idx_train, idx_valid) in enumerate(cv.split(X, y, groups=weeks_)):\n    #    if DEBUG:\n    #        print(idx)\n    #        print(idx_train.shape)\n    #        print(idx_valid.shape)\n    #    X_train, y_train, _ = prepare(X, idx_train)\n    #    if DEBUG:\n    #        print(f\"X_train shape is {X_train.shape}\")\n    #    X_valid, y_valid, dropeed_week_num = prepare(X, idx_valid)\n    #\n    #    current_model = model()\n    #    with SuppressOutput():\n    #        current_model.fit(X_train, y_train)\n    #    pred = current_model.predict(X_valid)\n    #    if DEBUG:\n    #        print(f\"pred mean {pred.mean()}\")\n    #    X_valid[\"WEEK_NUM\"] = dropeed_week_num\n    #    X_valid[\"target\"] = y_valid\n    #    X_valid[\"score\"] = pred\n    #    #X_valid = X_valid.with_columns(pl.Series(\"WEEK_NUM\", dropeed_week_num))\n    #    #X_valid = X_valid.with_columns(pl.Series(\"target\", y_valid))\n    #    #X_valid = X_valid.with_columns(pl.Series(\"score\", pred))\n    #    current_roc_auc_score = roc_auc_score(y_valid, pred)\n    #    if DEBUG:\n    #        print(f\"current_roc_auc_score is {current_roc_auc_score}\")\n    #    all_roc_scores.append(roc_auc_score(y_valid, pred))\n    #    try:\n    #        all_gini_scores.append(gini_stability(X_valid))\n    #    except ValueError:\n    #        print(\"could not calculate gini score\")\n    #    all_fitted_models.append(current_model)\n    #print(stability_results)\n    #print(np.array(stability_results).mean(axis=0))\n    return cv_results['test_score'], pd.concat([*stability_results]).groupby('fold').apply(gini_stability, include_groups=False).values, cv_results['estimator']","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.513669Z","iopub.status.idle":"2024-04-08T18:40:12.514023Z","shell.execute_reply.started":"2024-04-08T18:40:12.513852Z","shell.execute_reply":"2024-04-08T18:40:12.513866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_objective(X, y, get_model_function):\n    def objective(trial):\n        # Existing parameter suggestions\n        model = get_model_function(trial)\n        all_roc_scores, all_gini_scores, _ = infer(X, y, model)\n        \n        all_roc_score_mean = all_roc_scores.mean()\n        all_gini_score_mean = all_gini_scores.mean()\n        if DEBUG:\n            print(f\"all_roc_score_mean {all_roc_score_mean}\")\n            print(f\"all_gini_score_mean {all_gini_score_mean}\")\n        return all_roc_score_mean\n\n    return objective","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.515016Z","iopub.status.idle":"2024-04-08T18:40:12.515359Z","shell.execute_reply.started":"2024-04-08T18:40:12.515177Z","shell.execute_reply":"2024-04-08T18:40:12.515191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_config_from_study(model_key):\n    current_config = model_configs[model_key]\n    study = optuna.create_study(study_name=current_config[\"name\"], storage=current_config[\"storage\"], load_if_exists=True, direction='maximize')\n    return study.best_params","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.516995Z","iopub.status.idle":"2024-04-08T18:40:12.517325Z","shell.execute_reply.started":"2024-04-08T18:40:12.517146Z","shell.execute_reply":"2024-04-08T18:40:12.517159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_dataset():\n    frame = pre_process_frame(TEST_DIR, \"test\")\n    frame = frame.drop(\"WEEK_NUM\")\n    frame = frame.drop(\"index\")\n    return frame.to_pandas().fillna(-1)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.519043Z","iopub.status.idle":"2024-04-08T18:40:12.519457Z","shell.execute_reply.started":"2024-04-08T18:40:12.519221Z","shell.execute_reply":"2024-04-08T18:40:12.519236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_submission(prediction):\n    sample_submission = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/sample_submission.csv\")\n    sample_submission[\"score\"] = prediction\n    sample_submission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.520599Z","iopub.status.idle":"2024-04-08T18:40:12.520897Z","shell.execute_reply.started":"2024-04-08T18:40:12.520751Z","shell.execute_reply":"2024-04-08T18:40:12.520763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cv_value(model_key):\n    study = optuna.create_study(study_name=model_configs[model_key][\"name\"], storage=model_configs[model_key][\"storage\"], load_if_exists=True, direction='maximize')\n    return study.best_value","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.521879Z","iopub.status.idle":"2024-04-08T18:40:12.522205Z","shell.execute_reply.started":"2024-04-08T18:40:12.522026Z","shell.execute_reply":"2024-04-08T18:40:12.522038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_model(model):\n    best_model = model_configs[model]\n    study = optuna.create_study(study_name=best_model[\"name\"], storage=best_model[\"storage\"], load_if_exists=True, direction='maximize')\n    print(f\"study best params is {study.best_params}\")\n    print(f\"study best value is {study.best_value}\")\n    best_model = best_model[\"model_constructor\"](**study.best_params)\n    _, _, all_models = infer(X_inp, y_inp, best_model)\n            \n    df_test = get_test_dataset()\n    all_preds = []\n    for i in range(len(all_models)):\n        current_model = all_models[i]\n        model_configs[model][\"save_method\"](current_model, i)\n        all_preds.append(all_models[i].predict(df_test))\n    prediction = np.array(all_preds).mean(axis=0)\n    return prediction","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.523245Z","iopub.status.idle":"2024-04-08T18:40:12.523607Z","shell.execute_reply.started":"2024-04-08T18:40:12.523453Z","shell.execute_reply":"2024-04-08T18:40:12.523466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_and_pred(model, df_test):\n    all_preds = []\n    for fold in range(FOLDS):\n        current_model = model_configs[model][\"load_method\"](fold)\n        all_preds.append(current_model.predict(df_test))\n    return np.array(all_preds).mean(axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.524689Z","iopub.status.idle":"2024-04-08T18:40:12.525029Z","shell.execute_reply.started":"2024-04-08T18:40:12.524863Z","shell.execute_reply":"2024-04-08T18:40:12.524878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if OPTIMIZE_STACKING:\n    current_config = model_configs[\"Stacking\"]\n    study = optuna.create_study(study_name=current_config[\"name\"], storage=current_config[\"storage\"], load_if_exists=True, direction='maximize')\n    study.optimize(get_objective(X_inp, y_inp, \n                                 current_config[\"optuna_config\"]), \n                                 n_trials=TRIALS,\n                                 callbacks=[logging_callback])","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.525968Z","iopub.status.idle":"2024-04-08T18:40:12.526287Z","shell.execute_reply.started":"2024-04-08T18:40:12.526121Z","shell.execute_reply":"2024-04-08T18:40:12.526134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_to_ignore = [\"RF\", \"SVC\", \"NN\", \"Stacking\", \"Cat\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.527671Z","iopub.status.idle":"2024-04-08T18:40:12.527999Z","shell.execute_reply.started":"2024-04-08T18:40:12.527839Z","shell.execute_reply":"2024-04-08T18:40:12.527852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if OPTIMIZE:\n    for model in model_configs.keys():\n        if model in models_to_ignore:\n            print(f\"ignoring model {model}\")\n            continue\n        print(f\"optimizing {model}\")\n        current_train_model = model\n        current_config = model_configs[model]\n        study = optuna.create_study(study_name=current_config[\"name\"], storage=current_config[\"storage\"], load_if_exists=True, direction='maximize')\n        study.optimize(get_objective(X_inp, y_inp, \n                                 current_config[\"optuna_config\"]), \n                                 n_trials=TRIALS,\n                                 callbacks=[logging_callback])","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.530041Z","iopub.status.idle":"2024-04-08T18:40:12.530394Z","shell.execute_reply.started":"2024-04-08T18:40:12.530201Z","shell.execute_reply":"2024-04-08T18:40:12.530214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if INFER:\n    model_to_best_value = {}\n    if TRAIN_AND_SAVE_BEST_CONFIG_FOR_ONE_MODEL:\n        print(\"TRAIN_AND_SAVE_BEST_CONFIG_FOR_ONE_MODEL\")\n        prediction = predict_for_model(\"LGBM\")\n        generate_submission(prediction)\n        \n    if TRAIN_AND_SAVE_BEST_CONFIG:\n        print(\"TRAIN_AND_SAVE_BEST_CONFIG\")\n        for model in model_configs.keys():\n            if model in models_to_ignore:\n                continue\n            predict_for_model(model)\n    \n    if LOAD_AND_PREDICT:\n        print(\"LOAD_AND_PREDICT\")\n        df_test = get_test_dataset()\n        overall_preds = []\n        for model in model_configs.keys():\n            if model in models_to_ignore:\n                print(f\"ignoring model {model}\")\n                continue\n            best_value = get_cv_value(model)\n            print(f\"cross validation value is {best_value}\")\n            current_model_preds = []\n            for fold in range(FOLDS):    \n                current_model = model_configs[model][\"load_method\"](fold)\n                current_model_preds.append(current_model.predict(df_test))\n            overall_preds.append(np.array(current_model_preds).mean(axis=0))\n            generate_submission(*overall_preds)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:40:12.531397Z","iopub.status.idle":"2024-04-08T18:40:12.531718Z","shell.execute_reply.started":"2024-04-08T18:40:12.531564Z","shell.execute_reply":"2024-04-08T18:40:12.531577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if INFER_ENSEMBLE:\n    overall_preds = []\n    for model in model_configs.keys():\n        if model in models_to_ignore:\n            continue\n        print(f\"handling {model}\")\n        df_test = get_test_dataset()\n        model_prediction = load_and_pred(model, df_test)\n        overall_preds.append(model_prediction)\n    \n    #print(np.array(overall_preds))\n    mean_pred_for_models = np.array(overall_preds).mean(axis=0)\n    generate_submission(mean_pred_for_models)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:29:02.719086Z","iopub.status.idle":"2024-04-08T18:29:02.719610Z","shell.execute_reply.started":"2024-04-08T18:29:02.719404Z","shell.execute_reply":"2024-04-08T18:29:02.719419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if VALIDATE_ENSEMBLE:\n    cv = StratifiedGroupKFold(n_splits=FOLDS, shuffle=False,)\n    all_gini_stability_scores = []\n    for idx, (idx_train, idx_valid) in enumerate(cv.split(X_inp, y_inp, groups=weeks)):\n        current_overall_preds = []\n        for model in model_configs.keys():\n            if model in models_to_ignore:\n                continue\n            print(f\"handling {model}\")\n            \n            \n            df_test = X_inp.filter(pl.col(\"index\").is_in(idx_valid))\n            valid_target = df_test.select(pl.col(\"target\"))\n            df_test = df_test.drop(\"target\")\n            all_preds = load_and_pred(model, df_test)\n            current_overall_preds.append(all_preds)\n            #all_preds = []\n            #for fold in range(FOLDS):\n            #    current_model = model_configs[model][\"load_method\"](fold)\n            #    all_preds.append(current_model.predict(df_test.drop(\"WEEK_NUM\").to_pandas().fillna(-1)))\n            #current_overall_preds.append(np.array(all_preds).mean(axis=0))\n            \n            del all_preds; gc.collect()\n        \n        #print(current_overall_preds)\n        preds = np.array(current_overall_preds).mean(axis=0)\n        df_test = df_test.with_columns(pl.Series(name=\"score\", values=preds))\n        df_test = df_test.with_columns(pl.Series(\"target\", values=valid_target))\n        g_score = gini_stability(df_test)\n        print(f\"current g_score {g_score}\")\n        all_gini_stability_scores.append(g_score)\n        #print(gini_stability(df_test))\n    #print(f\"all_gini_stability_scores {all_gini_stability_scores}\")\n    mean_gini_stability_scores = np.array(all_gini_stability_scores).mean()\n    print(f\"mean_gini_stability_scores is {mean_gini_stability_scores}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T18:29:02.721121Z","iopub.status.idle":"2024-04-08T18:29:02.721439Z","shell.execute_reply.started":"2024-04-08T18:29:02.721285Z","shell.execute_reply":"2024-04-08T18:29:02.721299Z"},"trusted":true},"execution_count":null,"outputs":[]}]}