{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":7611702,"sourceType":"datasetVersion","datasetId":4432361},{"sourceId":7601505,"sourceType":"datasetVersion","datasetId":4425208,"isSourceIdPinned":false},{"sourceId":162438049,"sourceType":"kernelVersion"},{"sourceId":162470947,"sourceType":"kernelVersion"},{"sourceId":162602153,"sourceType":"kernelVersion"},{"sourceId":162605422,"sourceType":"kernelVersion"},{"sourceId":162611024,"sourceType":"kernelVersion"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction","metadata":{}},{"cell_type":"markdown","source":"**Related notebooks**\n\nUtility script notebook (with addtional functions, aggregators; data collection):\n\nhttps://www.kaggle.com/andreynesterov/home-credit-baseline-data\n\nTraining models notebooks:\n\nhttps://www.kaggle.com/andreynesterov/home-credit-baseline-training\n\nhttps://www.kaggle.com/code/andreynesterov/home-credit-baseline-training-no-dates\n\nhttps://www.kaggle.com/code/andreynesterov/home-credit-baseline-training-lightautoml","metadata":{}},{"cell_type":"markdown","source":"# Dependencies","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies pandas==2.0.3","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-02-12T14:12:04.816442Z","iopub.execute_input":"2024-02-12T14:12:04.817039Z","iopub.status.idle":"2024-02-12T14:12:29.934951Z","shell.execute_reply.started":"2024-02-12T14:12:04.817005Z","shell.execute_reply":"2024-02-12T14:12:29.934008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom sklearn.metrics import roc_auc_score, log_loss\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.preprocessing import LabelEncoder\nfrom scipy.optimize import minimize\n\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-12T14:12:29.936907Z","iopub.execute_input":"2024-02-12T14:12:29.937235Z","iopub.status.idle":"2024-02-12T14:12:34.576582Z","shell.execute_reply.started":"2024-02-12T14:12:29.937207Z","shell.execute_reply":"2024-02-12T14:12:34.575504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    model_1_data = Path(\"/kaggle/input/home-credit-baseline-training-all-feats\")\n    model_2_data = Path(\"/kaggle/input/d/batprem/home-credit-baseline-training-model-2\")\n    model_3_data = Path(\"/kaggle/input/home-credit-baseline-training-lightautoml\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:34.577760Z","iopub.execute_input":"2024-02-12T14:12:34.578047Z","iopub.status.idle":"2024-02-12T14:12:34.582455Z","shell.execute_reply.started":"2024-02-12T14:12:34.578023Z","shell.execute_reply":"2024-02-12T14:12:34.581583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"code","source":"import home_credit_baseline_data as data_nb","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:34.584836Z","iopub.execute_input":"2024-02-12T14:12:34.585154Z","iopub.status.idle":"2024-02-12T14:12:34.690254Z","shell.execute_reply.started":"2024-02-12T14:12:34.585130Z","shell.execute_reply":"2024-02-12T14:12:34.689580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### from https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts\n\ndef gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:34.691203Z","iopub.execute_input":"2024-02-12T14:12:34.691639Z","iopub.status.idle":"2024-02-12T14:12:34.698711Z","shell.execute_reply.started":"2024-02-12T14:12:34.691613Z","shell.execute_reply":"2024-02-12T14:12:34.697724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_proba_in_batches(model, data, batch_size=80000, predict_mode=\"base\"):\n    num_samples = len(data)\n    num_batches = int(np.ceil(num_samples / batch_size))\n    probabilities = np.zeros((num_samples,))\n\n    for batch_idx in range(num_batches):\n        print(f\"Processing batch: {batch_idx+1}/{num_batches}\")\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, num_samples)\n        X_batch = data.iloc[start_idx:end_idx]\n        if predict_mode == \"base\":\n            batch_probs = model.predict_proba(X_batch)[:, 1]\n        elif predict_mode == \"lightautoml\":\n            batch_probs = model.predict(X_batch).data.squeeze()\n        probabilities[start_idx:end_idx] = batch_probs\n        gc.collect()\n\n    return probabilities","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:34.699873Z","iopub.execute_input":"2024-02-12T14:12:34.700111Z","iopub.status.idle":"2024-02-12T14:12:34.710102Z","shell.execute_reply.started":"2024-02-12T14:12:34.700089Z","shell.execute_reply":"2024-02-12T14:12:34.709289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_df = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\")\ny_train = train_base_df[\"target\"]\noof_df = train_base_df\nmodels_score_df = pd.DataFrame()\ntest_preds_df = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:34.711370Z","iopub.execute_input":"2024-02-12T14:12:34.711653Z","iopub.status.idle":"2024-02-12T14:12:35.847802Z","shell.execute_reply.started":"2024-02-12T14:12:34.711630Z","shell.execute_reply":"2024-02-12T14:12:35.846862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 1","metadata":{}},{"cell_type":"code","source":"model_name = \"model_1\"\nmodel_1 = joblib.load(CFG.model_1_data / \"oof_model_1.pkl\")\nmodel_1","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:35.848866Z","iopub.execute_input":"2024-02-12T14:12:35.849146Z","iopub.status.idle":"2024-02-12T14:12:36.271446Z","shell.execute_reply.started":"2024-02-12T14:12:35.849121Z","shell.execute_reply":"2024-02-12T14:12:36.270572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(CFG.model_1_data / \"train_cat_columns.pkl\")\ntrain_cols = train_cols[~np.in1d(train_cols, \"target\")]\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:36.272687Z","iopub.execute_input":"2024-02-12T14:12:36.273374Z","iopub.status.idle":"2024-02-12T14:12:36.285385Z","shell.execute_reply.started":"2024-02-12T14:12:36.273337Z","shell.execute_reply":"2024-02-12T14:12:36.284540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_nb.Aggregator.group_aggregators = [pl.max, pl.min, pl.first, pl.last, pl.n_unique]\ntest_df = data_nb.prepare_df(data_nb.CFG.test_dir, cat_cols=cat_cols, mode=\"test\")\ndisplay(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:36.289579Z","iopub.execute_input":"2024-02-12T14:12:36.289918Z","iopub.status.idle":"2024-02-12T14:12:37.744995Z","shell.execute_reply.started":"2024-02-12T14:12:36.289893Z","shell.execute_reply":"2024-02-12T14:12:37.744116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds_df['case_id'] = test_df['case_id']\ntest_preds_df.set_index('case_id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:37.746095Z","iopub.execute_input":"2024-02-12T14:12:37.746363Z","iopub.status.idle":"2024-02-12T14:12:37.753944Z","shell.execute_reply.started":"2024-02-12T14:12:37.746338Z","shell.execute_reply":"2024-02-12T14:12:37.752885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df[train_cols].drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_1 = pd.Series(predict_proba_in_batches(model_1, X_test), index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_1","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:37.755152Z","iopub.execute_input":"2024-02-12T14:12:37.755557Z","iopub.status.idle":"2024-02-12T14:12:38.438146Z","shell.execute_reply.started":"2024-02-12T14:12:37.755510Z","shell.execute_reply":"2024-02-12T14:12:38.437156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(CFG.model_1_data / \"oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:38.439418Z","iopub.execute_input":"2024-02-12T14:12:38.440197Z","iopub.status.idle":"2024-02-12T14:12:38.600986Z","shell.execute_reply.started":"2024-02-12T14:12:38.440161Z","shell.execute_reply":"2024-02-12T14:12:38.600185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:38.602172Z","iopub.execute_input":"2024-02-12T14:12:38.602474Z","iopub.status.idle":"2024-02-12T14:12:39.386808Z","shell.execute_reply.started":"2024-02-12T14:12:38.602445Z","shell.execute_reply":"2024-02-12T14:12:39.385791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 2","metadata":{}},{"cell_type":"code","source":"model_name = \"model_2\"\nmodel_2 = joblib.load(CFG.model_2_data / \"oof_model.pkl\")\nmodel_2","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:39.388197Z","iopub.execute_input":"2024-02-12T14:12:39.388487Z","iopub.status.idle":"2024-02-12T14:12:39.735404Z","shell.execute_reply.started":"2024-02-12T14:12:39.388462Z","shell.execute_reply":"2024-02-12T14:12:39.734491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(CFG.model_2_data / \"train_cat_columns.pkl\")\ntrain_cols = train_cols[~np.in1d(train_cols, \"target\")]\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:39.736417Z","iopub.execute_input":"2024-02-12T14:12:39.736691Z","iopub.status.idle":"2024-02-12T14:12:39.748367Z","shell.execute_reply.started":"2024-02-12T14:12:39.736667Z","shell.execute_reply":"2024-02-12T14:12:39.747459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df[train_cols].drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_2 = pd.Series(predict_proba_in_batches(model_2, X_test), index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_2","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:39.749434Z","iopub.execute_input":"2024-02-12T14:12:39.749742Z","iopub.status.idle":"2024-02-12T14:12:40.421161Z","shell.execute_reply.started":"2024-02-12T14:12:39.749718Z","shell.execute_reply":"2024-02-12T14:12:40.420415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(CFG.model_2_data / \"oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:40.422238Z","iopub.execute_input":"2024-02-12T14:12:40.422583Z","iopub.status.idle":"2024-02-12T14:12:40.529898Z","shell.execute_reply.started":"2024-02-12T14:12:40.422550Z","shell.execute_reply":"2024-02-12T14:12:40.529131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:12:40.531015Z","iopub.execute_input":"2024-02-12T14:12:40.531302Z","iopub.status.idle":"2024-02-12T14:12:41.279177Z","shell.execute_reply.started":"2024-02-12T14:12:40.531277Z","shell.execute_reply":"2024-02-12T14:12:41.278269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 3","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies lightautoml==0.3.8","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-02-12T14:12:41.280389Z","iopub.execute_input":"2024-02-12T14:12:41.280771Z","iopub.status.idle":"2024-02-12T14:15:01.545240Z","shell.execute_reply.started":"2024-02-12T14:12:41.280736Z","shell.execute_reply":"2024-02-12T14:15:01.543718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:01.547033Z","iopub.execute_input":"2024-02-12T14:15:01.547438Z","iopub.status.idle":"2024-02-12T14:15:29.148171Z","shell.execute_reply.started":"2024-02-12T14:15:01.547398Z","shell.execute_reply":"2024-02-12T14:15:29.147377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = \"denselight_model\"\nmodel_3 = joblib.load(CFG.model_3_data / \"denselight_model.pkl\")\nmodel_3","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:29.149327Z","iopub.execute_input":"2024-02-12T14:15:29.150118Z","iopub.status.idle":"2024-02-12T14:15:29.961754Z","shell.execute_reply.started":"2024-02-12T14:15:29.150083Z","shell.execute_reply":"2024-02-12T14:15:29.960740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(CFG.model_3_data / \"train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:29.963101Z","iopub.execute_input":"2024-02-12T14:15:29.963457Z","iopub.status.idle":"2024-02-12T14:15:29.973879Z","shell.execute_reply.started":"2024-02-12T14:15:29.963425Z","shell.execute_reply":"2024-02-12T14:15:29.972968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_3 = pd.Series(\n    predict_proba_in_batches(model_3, X_test, predict_mode = \"lightautoml\"),\n    index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_3","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:29.974924Z","iopub.execute_input":"2024-02-12T14:15:29.975281Z","iopub.status.idle":"2024-02-12T14:15:33.313524Z","shell.execute_reply.started":"2024-02-12T14:15:29.975257Z","shell.execute_reply":"2024-02-12T14:15:33.312757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(CFG.model_3_data / \"denselight_oof_preds.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:33.314552Z","iopub.execute_input":"2024-02-12T14:15:33.314824Z","iopub.status.idle":"2024-02-12T14:15:33.373834Z","shell.execute_reply.started":"2024-02-12T14:15:33.314800Z","shell.execute_reply":"2024-02-12T14:15:33.373063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:33.375390Z","iopub.execute_input":"2024-02-12T14:15:33.375690Z","iopub.status.idle":"2024-02-12T14:15:34.200553Z","shell.execute_reply.started":"2024-02-12T14:15:33.375664Z","shell.execute_reply":"2024-02-12T14:15:34.199560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Estimation Results","metadata":{}},{"cell_type":"code","source":"models_score_df","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:34.202167Z","iopub.execute_input":"2024-02-12T14:15:34.202512Z","iopub.status.idle":"2024-02-12T14:15:34.210671Z","shell.execute_reply.started":"2024-02-12T14:15:34.202480Z","shell.execute_reply":"2024-02-12T14:15:34.209661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:34.217074Z","iopub.execute_input":"2024-02-12T14:15:34.217360Z","iopub.status.idle":"2024-02-12T14:15:34.237167Z","shell.execute_reply.started":"2024-02-12T14:15:34.217336Z","shell.execute_reply":"2024-02-12T14:15:34.236226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_df\ngc.collect()","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-12T14:15:34.238429Z","iopub.execute_input":"2024-02-12T14:15:34.238831Z","iopub.status.idle":"2024-02-12T14:15:34.490554Z","shell.execute_reply.started":"2024-02-12T14:15:34.238797Z","shell.execute_reply":"2024-02-12T14:15:34.489599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Blending","metadata":{}},{"cell_type":"code","source":"def gini_wrapper(base_df):\n    base_df = base_df[[\"WEEK_NUM\", \"target\"]].copy()\n    def gini_wrapper_inner(target, scores):\n        base_df[\"score\"] = scores\n        gini_score = gini_stability(base_df, score_col=\"score\")\n        return 1 - gini_score\n    return gini_wrapper_inner","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:34.491907Z","iopub.execute_input":"2024-02-12T14:15:34.492342Z","iopub.status.idle":"2024-02-12T14:15:34.500071Z","shell.execute_reply.started":"2024-02-12T14:15:34.492308Z","shell.execute_reply":"2024-02-12T14:15:34.499227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hill climbing using minimize","metadata":{}},{"cell_type":"code","source":"class WeightsSearcher:\n    def __init__(self, loss_fn, bounds=[], mode=\"min\", method='SLSQP'):\n        self.loss_fn = loss_fn\n        self.bounds = bounds\n        self.mode = mode\n        self.method = method # Nelder-Mead - for not smooth functions\n        \n    def _objective_function_wrapper(self, pred_values, true_targets, obj_fn):\n        def objective_function(weights):\n            pred_weighted = (pred_values * weights).sum(axis=1)\n            score = obj_fn(true_targets, pred_weighted)\n            return score\n        return objective_function\n    \n    def find_weights(self, val_preds, true_targets):\n        len_models = len(self.bounds)\n        bounds = [0,1] * len_models if len(self.bounds) == 0 else self.bounds\n        initial_weights = np.ones(len_models) / len_models\n        objective_function = self._objective_function_wrapper(val_preds, true_targets, self.loss_fn)\n        result = minimize(\n            objective_function, \n            initial_weights, \n            bounds=bounds, \n            method=self.method,\n        )\n        optimized_weights = result.x\n        optimized_weights /= np.sum(optimized_weights)\n        return optimized_weights","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:34.501324Z","iopub.execute_input":"2024-02-12T14:15:34.501735Z","iopub.status.idle":"2024-02-12T14:15:34.512212Z","shell.execute_reply.started":"2024-02-12T14:15:34.501700Z","shell.execute_reply":"2024-02-12T14:15:34.511301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_names = models_score_df.index.to_list()\npred_cols = [f\"pred_{name}\" for name in model_names]\nmodel_names","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:34.513519Z","iopub.execute_input":"2024-02-12T14:15:34.513864Z","iopub.status.idle":"2024-02-12T14:15:34.526090Z","shell.execute_reply.started":"2024-02-12T14:15:34.513841Z","shell.execute_reply":"2024-02-12T14:15:34.525186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bounds = [(0, 1)] * len(pred_cols)\n# roc_auc_fn = lambda y_true, y_pred: 1 - roc_auc_score(y_true, y_pred)\ngini_score_fn = gini_wrapper(oof_df)\nw_searcher = WeightsSearcher(gini_score_fn, bounds, method='Nelder-Mead') # log_loss, gini_stability, roc_auc_fn\noptimized_weights = w_searcher.find_weights(\n    oof_df[pred_cols].to_numpy(), \n    y_train\n)\noptimized_weights_df = pd.DataFrame(zip(model_names, optimized_weights), columns=['model', 'weight'])\ndisplay(optimized_weights_df)\nprint(\"sum: \", np.sum(optimized_weights))","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:15:34.527270Z","iopub.execute_input":"2024-02-12T14:15:34.527600Z","iopub.status.idle":"2024-02-12T14:16:48.086146Z","shell.execute_reply.started":"2024-02-12T14:15:34.527569Z","shell.execute_reply":"2024-02-12T14:16:48.085095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_pred_optimized = (oof_df[pred_cols] * optimized_weights).sum(axis=1).to_numpy()\noof_df[\"pred_optimized\"] = oof_pred_optimized\nroc_auc_oof = roc_auc_score(y_train, oof_pred_optimized)\ngini_score = gini_stability(oof_df, score_col=\"pred_optimized\")\nprint(\"CV roc_auc_oof optimized:\\t\", roc_auc_oof)\nprint(\"CV gini_score:\\t\\t\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:16:48.087468Z","iopub.execute_input":"2024-02-12T14:16:48.087855Z","iopub.status.idle":"2024-02-12T14:16:50.067669Z","shell.execute_reply.started":"2024-02-12T14:16:48.087821Z","shell.execute_reply":"2024-02-12T14:16:50.066589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:16:50.068986Z","iopub.execute_input":"2024-02-12T14:16:50.069793Z","iopub.status.idle":"2024-02-12T14:16:50.082063Z","shell.execute_reply.started":"2024-02-12T14:16:50.069765Z","shell.execute_reply":"2024-02-12T14:16:50.081014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_all = (optimized_weights * test_preds_df[pred_cols]).sum(axis=1).to_numpy()\ny_pred_all[:10]","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:16:50.083409Z","iopub.execute_input":"2024-02-12T14:16:50.083799Z","iopub.status.idle":"2024-02-12T14:16:50.098680Z","shell.execute_reply.started":"2024-02-12T14:16:50.083766Z","shell.execute_reply":"2024-02-12T14:16:50.097796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"subm_df = pd.read_csv(data_nb.CFG.root_dir / \"sample_submission.csv\")\nsubm_df = subm_df.set_index(\"case_id\")\nsubm_df[\"score\"] = y_pred_all\ndisplay(subm_df.head())\nprint(\"Check null: \", subm_df[\"score\"].isnull().any())","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:16:50.099684Z","iopub.execute_input":"2024-02-12T14:16:50.099979Z","iopub.status.idle":"2024-02-12T14:16:50.123364Z","shell.execute_reply.started":"2024-02-12T14:16:50.099946Z","shell.execute_reply":"2024-02-12T14:16:50.122443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T14:16:50.124665Z","iopub.execute_input":"2024-02-12T14:16:50.125378Z","iopub.status.idle":"2024-02-12T14:16:50.131796Z","shell.execute_reply.started":"2024-02-12T14:16:50.125342Z","shell.execute_reply":"2024-02-12T14:16:50.130962Z"},"trusted":true},"execution_count":null,"outputs":[]}]}