{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":162363627,"sourceType":"kernelVersion"},{"sourceId":162364554,"sourceType":"kernelVersion"},{"sourceId":162403520,"sourceType":"kernelVersion"},{"sourceId":162438049,"sourceType":"kernelVersion"},{"sourceId":162470947,"sourceType":"kernelVersion"}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction","metadata":{}},{"cell_type":"markdown","source":"**Related notebooks**\n\nUtility script notebook (with addtional functions, aggregators; data collection):\n\nhttps://www.kaggle.com/andreynesterov/home-credit-baseline-data\n\nTraining models notebooks:\n\nhttps://www.kaggle.com/andreynesterov/home-credit-baseline-training\n\nhttps://www.kaggle.com/code/andreynesterov/home-credit-baseline-training-no-dates\n\nhttps://www.kaggle.com/code/andreynesterov/home-credit-baseline-training-lightautoml","metadata":{}},{"cell_type":"markdown","source":"# Dependencies","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies pandas==2.0.3","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-02-11T08:24:45.332316Z","iopub.execute_input":"2024-02-11T08:24:45.332687Z","iopub.status.idle":"2024-02-11T08:25:10.362952Z","shell.execute_reply.started":"2024-02-11T08:24:45.332656Z","shell.execute_reply":"2024-02-11T08:25:10.361856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom sklearn.metrics import roc_auc_score, log_loss\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.preprocessing import LabelEncoder\nfrom scipy.optimize import minimize\n\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-11T08:25:10.364874Z","iopub.execute_input":"2024-02-11T08:25:10.365149Z","iopub.status.idle":"2024-02-11T08:25:15.22946Z","shell.execute_reply.started":"2024-02-11T08:25:10.365124Z","shell.execute_reply":"2024-02-11T08:25:15.228286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"code","source":"import home_credit_baseline_data as data_nb","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:15.230962Z","iopub.execute_input":"2024-02-11T08:25:15.231645Z","iopub.status.idle":"2024-02-11T08:25:15.324829Z","shell.execute_reply.started":"2024-02-11T08:25:15.231588Z","shell.execute_reply":"2024-02-11T08:25:15.324026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### from https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts\n\ndef gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:15.327536Z","iopub.execute_input":"2024-02-11T08:25:15.328106Z","iopub.status.idle":"2024-02-11T08:25:15.335321Z","shell.execute_reply.started":"2024-02-11T08:25:15.328071Z","shell.execute_reply":"2024-02-11T08:25:15.334319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_proba_in_batches(model, data, batch_size=100000, predict_mode=\"base\"):\n    num_samples = len(data)\n    num_batches = int(np.ceil(num_samples / batch_size))\n    probabilities = np.zeros((num_samples,))\n\n    for batch_idx in range(num_batches):\n        print(f\"Processing batch: {batch_idx+1}/{num_batches}\")\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, num_samples)\n        X_batch = data.iloc[start_idx:end_idx]\n        if predict_mode == \"base\":\n            batch_probs = model.predict_proba(X_batch)[:, 1]\n        elif predict_mode == \"lightautoml\":\n            batch_probs = model.predict(X_batch).data.squeeze()\n        probabilities[start_idx:end_idx] = batch_probs\n        gc.collect()\n\n    return probabilities","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:15.336434Z","iopub.execute_input":"2024-02-11T08:25:15.336744Z","iopub.status.idle":"2024-02-11T08:25:15.346756Z","shell.execute_reply.started":"2024-02-11T08:25:15.33672Z","shell.execute_reply":"2024-02-11T08:25:15.345965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_df = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\")\ny_train = train_base_df[\"target\"]\noof_df = train_base_df\nmodels_score_df = pd.DataFrame()\n# model_names = []\ntest_preds_df = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:15.347759Z","iopub.execute_input":"2024-02-11T08:25:15.348076Z","iopub.status.idle":"2024-02-11T08:25:16.509056Z","shell.execute_reply.started":"2024-02-11T08:25:15.348044Z","shell.execute_reply":"2024-02-11T08:25:16.508257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 1","metadata":{}},{"cell_type":"code","source":"model_name = \"model_1\"\nmodel_1 = joblib.load(\"/kaggle/input/home-credit-baseline-training/oof_model_1.pkl\")\nmodel_1\n# model_names.append(model_name)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:16.51033Z","iopub.execute_input":"2024-02-11T08:25:16.510709Z","iopub.status.idle":"2024-02-11T08:25:16.873089Z","shell.execute_reply.started":"2024-02-11T08:25:16.510677Z","shell.execute_reply":"2024-02-11T08:25:16.872155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:16.874311Z","iopub.execute_input":"2024-02-11T08:25:16.874619Z","iopub.status.idle":"2024-02-11T08:25:16.883904Z","shell.execute_reply.started":"2024-02-11T08:25:16.874582Z","shell.execute_reply":"2024-02-11T08:25:16.88302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = data_nb.prepare_df(data_nb.CFG.test_dir, cat_cols=cat_cols, mode=\"test\", train_cols=train_cols)\ndisplay(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:16.8851Z","iopub.execute_input":"2024-02-11T08:25:16.885443Z","iopub.status.idle":"2024-02-11T08:25:18.083413Z","shell.execute_reply.started":"2024-02-11T08:25:16.88541Z","shell.execute_reply":"2024-02-11T08:25:18.082366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds_df['case_id'] = test_df['case_id']\ntest_preds_df.set_index('case_id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:18.087944Z","iopub.execute_input":"2024-02-11T08:25:18.088615Z","iopub.status.idle":"2024-02-11T08:25:18.09659Z","shell.execute_reply.started":"2024-02-11T08:25:18.088568Z","shell.execute_reply":"2024-02-11T08:25:18.095573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_1 = pd.Series(predict_proba_in_batches(model_1, X_test), index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_1","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:18.097808Z","iopub.execute_input":"2024-02-11T08:25:18.098117Z","iopub.status.idle":"2024-02-11T08:25:18.825227Z","shell.execute_reply.started":"2024-02-11T08:25:18.098092Z","shell.execute_reply":"2024-02-11T08:25:18.824423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training/oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:18.826296Z","iopub.execute_input":"2024-02-11T08:25:18.826589Z","iopub.status.idle":"2024-02-11T08:25:18.95658Z","shell.execute_reply.started":"2024-02-11T08:25:18.826558Z","shell.execute_reply":"2024-02-11T08:25:18.955751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:18.957723Z","iopub.execute_input":"2024-02-11T08:25:18.957999Z","iopub.status.idle":"2024-02-11T08:25:19.809832Z","shell.execute_reply.started":"2024-02-11T08:25:18.957976Z","shell.execute_reply":"2024-02-11T08:25:19.808822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 2","metadata":{}},{"cell_type":"code","source":"model_name = \"model_2\"\nmodel_2 = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/oof_model.pkl\")\nmodel_2","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:19.810953Z","iopub.execute_input":"2024-02-11T08:25:19.811236Z","iopub.status.idle":"2024-02-11T08:25:20.18105Z","shell.execute_reply.started":"2024-02-11T08:25:19.811212Z","shell.execute_reply":"2024-02-11T08:25:20.180088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:20.182201Z","iopub.execute_input":"2024-02-11T08:25:20.182489Z","iopub.status.idle":"2024-02-11T08:25:20.194795Z","shell.execute_reply.started":"2024-02-11T08:25:20.182464Z","shell.execute_reply":"2024-02-11T08:25:20.193764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_2 = pd.Series(predict_proba_in_batches(model_2, X_test), index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_2","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:20.195841Z","iopub.execute_input":"2024-02-11T08:25:20.1961Z","iopub.status.idle":"2024-02-11T08:25:20.90744Z","shell.execute_reply.started":"2024-02-11T08:25:20.196078Z","shell.execute_reply":"2024-02-11T08:25:20.906371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:20.908819Z","iopub.execute_input":"2024-02-11T08:25:20.90911Z","iopub.status.idle":"2024-02-11T08:25:21.005416Z","shell.execute_reply.started":"2024-02-11T08:25:20.909085Z","shell.execute_reply":"2024-02-11T08:25:21.004649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:25:21.006525Z","iopub.execute_input":"2024-02-11T08:25:21.007229Z","iopub.status.idle":"2024-02-11T08:25:21.854834Z","shell.execute_reply.started":"2024-02-11T08:25:21.0072Z","shell.execute_reply":"2024-02-11T08:25:21.853917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 3","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies lightautoml==0.3.8","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-02-11T08:25:21.856068Z","iopub.execute_input":"2024-02-11T08:25:21.856849Z","iopub.status.idle":"2024-02-11T08:27:39.217109Z","shell.execute_reply.started":"2024-02-11T08:25:21.856811Z","shell.execute_reply":"2024-02-11T08:27:39.215803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:27:39.218851Z","iopub.execute_input":"2024-02-11T08:27:39.219179Z","iopub.status.idle":"2024-02-11T08:28:09.912847Z","shell.execute_reply.started":"2024-02-11T08:27:39.219149Z","shell.execute_reply":"2024-02-11T08:28:09.911835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = \"denselight_model\"\nmodel_3 = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/denselight_model.pkl\")\nmodel_3","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:09.914057Z","iopub.execute_input":"2024-02-11T08:28:09.914739Z","iopub.status.idle":"2024-02-11T08:28:10.970865Z","shell.execute_reply.started":"2024-02-11T08:28:09.914711Z","shell.execute_reply":"2024-02-11T08:28:10.969878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:10.972004Z","iopub.execute_input":"2024-02-11T08:28:10.972282Z","iopub.status.idle":"2024-02-11T08:28:10.982249Z","shell.execute_reply.started":"2024-02-11T08:28:10.972259Z","shell.execute_reply":"2024-02-11T08:28:10.981369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_3 = pd.Series(\n    predict_proba_in_batches(model_3, X_test, predict_mode = \"lightautoml\"),\n    index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_3","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:10.983477Z","iopub.execute_input":"2024-02-11T08:28:10.984037Z","iopub.status.idle":"2024-02-11T08:28:14.540954Z","shell.execute_reply.started":"2024-02-11T08:28:10.984013Z","shell.execute_reply":"2024-02-11T08:28:14.540173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/denselight_oof_preds.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:14.542406Z","iopub.execute_input":"2024-02-11T08:28:14.542768Z","iopub.status.idle":"2024-02-11T08:28:14.628093Z","shell.execute_reply.started":"2024-02-11T08:28:14.542735Z","shell.execute_reply":"2024-02-11T08:28:14.627211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:14.629218Z","iopub.execute_input":"2024-02-11T08:28:14.629509Z","iopub.status.idle":"2024-02-11T08:28:15.429448Z","shell.execute_reply.started":"2024-02-11T08:28:14.629485Z","shell.execute_reply":"2024-02-11T08:28:15.428525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Estimation Results","metadata":{}},{"cell_type":"code","source":"models_score_df","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:15.449513Z","iopub.execute_input":"2024-02-11T08:28:15.449852Z","iopub.status.idle":"2024-02-11T08:28:15.461526Z","shell.execute_reply.started":"2024-02-11T08:28:15.449826Z","shell.execute_reply":"2024-02-11T08:28:15.460521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:15.430672Z","iopub.execute_input":"2024-02-11T08:28:15.431524Z","iopub.status.idle":"2024-02-11T08:28:15.448333Z","shell.execute_reply.started":"2024-02-11T08:28:15.431484Z","shell.execute_reply":"2024-02-11T08:28:15.447324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_df\ngc.collect()","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-11T08:28:15.466992Z","iopub.execute_input":"2024-02-11T08:28:15.46734Z","iopub.status.idle":"2024-02-11T08:28:15.744749Z","shell.execute_reply.started":"2024-02-11T08:28:15.467316Z","shell.execute_reply":"2024-02-11T08:28:15.743691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Blending","metadata":{}},{"cell_type":"code","source":"def gini_wrapper(base_df):\n    base_df = base_df[[\"WEEK_NUM\", \"target\"]].copy()\n    def gini_wrapper_inner(target, scores):\n        base_df[\"score\"] = scores\n        gini_score = gini_stability(base_df, score_col=\"score\")\n        return 1 - gini_score\n    return gini_wrapper_inner","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:15.746044Z","iopub.execute_input":"2024-02-11T08:28:15.746406Z","iopub.status.idle":"2024-02-11T08:28:15.753561Z","shell.execute_reply.started":"2024-02-11T08:28:15.746375Z","shell.execute_reply":"2024-02-11T08:28:15.752803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hill climbing using minimize","metadata":{}},{"cell_type":"code","source":"class WeightsSearcher:\n    def __init__(self, loss_fn, bounds=[], mode=\"min\", method='SLSQP'):\n        self.loss_fn = loss_fn\n        self.bounds = bounds\n        self.mode = mode\n        self.method = method # Nelder-Mead - for not smooth functions\n        \n    def _objective_function_wrapper(self, pred_values, true_targets, obj_fn):\n        def objective_function(weights):\n            pred_weighted = (pred_values * weights).sum(axis=1)\n            score = obj_fn(true_targets, pred_weighted)\n            return score\n        return objective_function\n    \n    def find_weights(self, val_preds, true_targets):\n        len_models = len(self.bounds)\n        bounds = [0,1] * len_models if len(self.bounds) == 0 else self.bounds\n        initial_weights = np.ones(len_models) / len_models\n        objective_function = self._objective_function_wrapper(val_preds, true_targets, self.loss_fn)\n        result = minimize(\n            objective_function, \n            initial_weights, \n            bounds=bounds, \n            method=self.method,\n        )\n        optimized_weights = result.x\n        optimized_weights /= np.sum(optimized_weights)\n        return optimized_weights","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:15.75481Z","iopub.execute_input":"2024-02-11T08:28:15.755158Z","iopub.status.idle":"2024-02-11T08:28:15.765437Z","shell.execute_reply.started":"2024-02-11T08:28:15.755128Z","shell.execute_reply":"2024-02-11T08:28:15.764714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_names = models_score_df.index.to_list()\npred_cols = [f\"pred_{name}\" for name in model_names]\nmodel_names","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:28:15.766586Z","iopub.execute_input":"2024-02-11T08:28:15.766891Z","iopub.status.idle":"2024-02-11T08:28:15.777934Z","shell.execute_reply.started":"2024-02-11T08:28:15.766868Z","shell.execute_reply":"2024-02-11T08:28:15.777085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bounds = [(0, 1)] * len(pred_cols)\nroc_auc_fn = lambda y_true, y_pred: 1 - roc_auc_score(y_true, y_pred)\ngini_score_fn = gini_wrapper(oof_df)\nw_searcher = WeightsSearcher(gini_score_fn, bounds, method='Nelder-Mead') # log_loss, gini_stability roc_auc_fn\noptimized_weights = w_searcher.find_weights(\n    oof_df[pred_cols].to_numpy(), \n    y_train\n)\noptimized_weights_df = pd.DataFrame(zip(model_names, optimized_weights), columns=['model', 'weight'])\ndisplay(optimized_weights_df)\nprint(\"sum: \", np.sum(optimized_weights))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_pred_optimized = (oof_df[pred_cols] * optimized_weights).sum(axis=1).to_numpy()\noof_df[\"pred_optimized\"] = oof_pred_optimized\nroc_auc_oof = roc_auc_score(y_train, oof_pred_optimized)\ngini_score = gini_stability(oof_df, score_col=\"pred_optimized\")\nprint(\"CV roc_auc_oof optimized:\\t\", roc_auc_oof)\nprint(\"CV gini_score:\\t\\t\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:29:25.358993Z","iopub.execute_input":"2024-02-11T08:29:25.359324Z","iopub.status.idle":"2024-02-11T08:29:27.335487Z","shell.execute_reply.started":"2024-02-11T08:29:25.359297Z","shell.execute_reply":"2024-02-11T08:29:27.334416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_all = (optimized_weights * test_preds_df[pred_cols]).sum(axis=1).to_numpy()\ny_pred_all[:10]","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:29:27.336535Z","iopub.execute_input":"2024-02-11T08:29:27.336809Z","iopub.status.idle":"2024-02-11T08:29:27.346352Z","shell.execute_reply.started":"2024-02-11T08:29:27.336786Z","shell.execute_reply":"2024-02-11T08:29:27.34527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"subm_df = pd.read_csv(data_nb.CFG.root_dir / \"sample_submission.csv\")\nsubm_df = subm_df.set_index(\"case_id\")\nsubm_df[\"score\"] = y_pred_all\ndisplay(subm_df.head())\nprint(\"Check null: \", subm_df[\"score\"].isnull().any())","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:29:27.347547Z","iopub.execute_input":"2024-02-11T08:29:27.347861Z","iopub.status.idle":"2024-02-11T08:29:27.372231Z","shell.execute_reply.started":"2024-02-11T08:29:27.347838Z","shell.execute_reply":"2024-02-11T08:29:27.371347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T08:29:27.373635Z","iopub.execute_input":"2024-02-11T08:29:27.374263Z","iopub.status.idle":"2024-02-11T08:29:27.380686Z","shell.execute_reply.started":"2024-02-11T08:29:27.374226Z","shell.execute_reply":"2024-02-11T08:29:27.379818Z"},"trusted":true},"execution_count":null,"outputs":[]}]}