{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":162438049,"sourceType":"kernelVersion"},{"sourceId":162470947,"sourceType":"kernelVersion"},{"sourceId":162602153,"sourceType":"kernelVersion"},{"sourceId":162605422,"sourceType":"kernelVersion"},{"sourceId":162611024,"sourceType":"kernelVersion"}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is an experiment to see the effect of [the metric's trick](https://www.kaggle.com/competitions/home-credit-credit-risk-model-stability/discussion/476449) on one of the top-scoring [public notebooks](https://www.kaggle.com/code/andreynesterov/home-credit-baseline-inference). As the result, the public LB changed from 0.575 to 0.605, i.e. by the same ~0.03 as it was reported by @at7459. If you find this notebook useful, please take your time to appreciate the original work by @andreynesterov.","metadata":{}},{"cell_type":"markdown","source":"# Introduction\n\nRelated notebooks\n\nUtility script notebook (with addtional functions, aggregators; data collection):\n\nhttps://www.kaggle.com/andreynesterov/home-credit-baseline-data\n\nTraining models notebooks:\n\nhttps://www.kaggle.com/andreynesterov/home-credit-baseline-training\n\nhttps://www.kaggle.com/code/andreynesterov/home-credit-baseline-training-no-dates\n\nhttps://www.kaggle.com/code/andreynesterov/home-credit-baseline-training-lightautoml","metadata":{}},{"cell_type":"markdown","source":"# Dependencies","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies pandas==2.0.3","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-02-13T02:35:12.784861Z","iopub.execute_input":"2024-02-13T02:35:12.785490Z","iopub.status.idle":"2024-02-13T02:35:25.091880Z","shell.execute_reply.started":"2024-02-13T02:35:12.785438Z","shell.execute_reply":"2024-02-13T02:35:25.090563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom sklearn.metrics import roc_auc_score, log_loss\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.preprocessing import LabelEncoder\nfrom scipy.optimize import minimize\n\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-13T02:35:25.094775Z","iopub.execute_input":"2024-02-13T02:35:25.095205Z","iopub.status.idle":"2024-02-13T02:35:25.102321Z","shell.execute_reply.started":"2024-02-13T02:35:25.095167Z","shell.execute_reply":"2024-02-13T02:35:25.101345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"code","source":"import home_credit_baseline_data as data_nb","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:25.103678Z","iopub.execute_input":"2024-02-13T02:35:25.104039Z","iopub.status.idle":"2024-02-13T02:35:25.151114Z","shell.execute_reply.started":"2024-02-13T02:35:25.104005Z","shell.execute_reply":"2024-02-13T02:35:25.150119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### from https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts\n\ndef gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:25.153520Z","iopub.execute_input":"2024-02-13T02:35:25.153848Z","iopub.status.idle":"2024-02-13T02:35:25.161698Z","shell.execute_reply.started":"2024-02-13T02:35:25.153815Z","shell.execute_reply":"2024-02-13T02:35:25.160664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_proba_in_batches(model, data, batch_size=100000, predict_mode=\"base\"):\n    num_samples = len(data)\n    num_batches = int(np.ceil(num_samples / batch_size))\n    probabilities = np.zeros((num_samples,))\n\n    for batch_idx in range(num_batches):\n        print(f\"Processing batch: {batch_idx+1}/{num_batches}\")\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, num_samples)\n        X_batch = data.iloc[start_idx:end_idx]\n        if predict_mode == \"base\":\n            batch_probs = model.predict_proba(X_batch)[:, 1]\n        elif predict_mode == \"lightautoml\":\n            batch_probs = model.predict(X_batch).data.squeeze()\n        probabilities[start_idx:end_idx] = batch_probs\n        gc.collect()\n\n    return probabilities","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:25.162918Z","iopub.execute_input":"2024-02-13T02:35:25.163210Z","iopub.status.idle":"2024-02-13T02:35:25.173233Z","shell.execute_reply.started":"2024-02-13T02:35:25.163187Z","shell.execute_reply":"2024-02-13T02:35:25.172442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_df = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\")\ny_train = train_base_df[\"target\"]\noof_df = train_base_df\nmodels_score_df = pd.DataFrame()\n# model_names = []\ntest_preds_df = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:25.174571Z","iopub.execute_input":"2024-02-13T02:35:25.175214Z","iopub.status.idle":"2024-02-13T02:35:25.928660Z","shell.execute_reply.started":"2024-02-13T02:35:25.175182Z","shell.execute_reply":"2024-02-13T02:35:25.927853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 1","metadata":{}},{"cell_type":"code","source":"model_name = \"model_1\"\nmodel_1 = joblib.load(\"/kaggle/input/home-credit-baseline-training/oof_model_1.pkl\")\nmodel_1\n# model_names.append(model_name)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:25.929989Z","iopub.execute_input":"2024-02-13T02:35:25.930464Z","iopub.status.idle":"2024-02-13T02:35:26.142481Z","shell.execute_reply.started":"2024-02-13T02:35:25.930430Z","shell.execute_reply":"2024-02-13T02:35:26.141577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:26.143831Z","iopub.execute_input":"2024-02-13T02:35:26.144212Z","iopub.status.idle":"2024-02-13T02:35:26.152544Z","shell.execute_reply.started":"2024-02-13T02:35:26.144177Z","shell.execute_reply":"2024-02-13T02:35:26.151658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = data_nb.prepare_df(data_nb.CFG.test_dir, cat_cols=cat_cols, mode=\"test\", train_cols=train_cols)\ndisplay(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:26.153698Z","iopub.execute_input":"2024-02-13T02:35:26.154564Z","iopub.status.idle":"2024-02-13T02:35:27.595916Z","shell.execute_reply.started":"2024-02-13T02:35:26.154531Z","shell.execute_reply":"2024-02-13T02:35:27.595037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds_df['case_id'] = test_df['case_id']\ntest_preds_df.set_index('case_id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:27.600118Z","iopub.execute_input":"2024-02-13T02:35:27.600525Z","iopub.status.idle":"2024-02-13T02:35:27.607059Z","shell.execute_reply.started":"2024-02-13T02:35:27.600495Z","shell.execute_reply":"2024-02-13T02:35:27.605997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_1 = pd.Series(predict_proba_in_batches(model_1, X_test), index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_1","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:27.608183Z","iopub.execute_input":"2024-02-13T02:35:27.608766Z","iopub.status.idle":"2024-02-13T02:35:28.758271Z","shell.execute_reply.started":"2024-02-13T02:35:27.608731Z","shell.execute_reply":"2024-02-13T02:35:28.757371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training/oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:28.759421Z","iopub.execute_input":"2024-02-13T02:35:28.759717Z","iopub.status.idle":"2024-02-13T02:35:28.773225Z","shell.execute_reply.started":"2024-02-13T02:35:28.759694Z","shell.execute_reply":"2024-02-13T02:35:28.772351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:28.774452Z","iopub.execute_input":"2024-02-13T02:35:28.774923Z","iopub.status.idle":"2024-02-13T02:35:29.547952Z","shell.execute_reply.started":"2024-02-13T02:35:28.774887Z","shell.execute_reply":"2024-02-13T02:35:29.547018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 2","metadata":{}},{"cell_type":"code","source":"model_name = \"model_2\"\nmodel_2 = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/oof_model.pkl\")\nmodel_2","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:29.549216Z","iopub.execute_input":"2024-02-13T02:35:29.549531Z","iopub.status.idle":"2024-02-13T02:35:29.772950Z","shell.execute_reply.started":"2024-02-13T02:35:29.549505Z","shell.execute_reply":"2024-02-13T02:35:29.772036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:29.774025Z","iopub.execute_input":"2024-02-13T02:35:29.774360Z","iopub.status.idle":"2024-02-13T02:35:29.781395Z","shell.execute_reply.started":"2024-02-13T02:35:29.774335Z","shell.execute_reply":"2024-02-13T02:35:29.780574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_2 = pd.Series(predict_proba_in_batches(model_2, X_test), index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_2","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:29.782654Z","iopub.execute_input":"2024-02-13T02:35:29.783201Z","iopub.status.idle":"2024-02-13T02:35:30.929967Z","shell.execute_reply.started":"2024-02-13T02:35:29.783168Z","shell.execute_reply":"2024-02-13T02:35:30.928931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/oof_pred.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:30.931159Z","iopub.execute_input":"2024-02-13T02:35:30.931467Z","iopub.status.idle":"2024-02-13T02:35:30.944795Z","shell.execute_reply.started":"2024-02-13T02:35:30.931443Z","shell.execute_reply":"2024-02-13T02:35:30.944031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:30.945812Z","iopub.execute_input":"2024-02-13T02:35:30.946111Z","iopub.status.idle":"2024-02-13T02:35:31.756190Z","shell.execute_reply.started":"2024-02-13T02:35:30.946073Z","shell.execute_reply":"2024-02-13T02:35:31.755191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 3","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies lightautoml==0.3.8","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-02-13T02:35:31.757459Z","iopub.execute_input":"2024-02-13T02:35:31.757781Z","iopub.status.idle":"2024-02-13T02:35:49.819416Z","shell.execute_reply.started":"2024-02-13T02:35:31.757754Z","shell.execute_reply":"2024-02-13T02:35:49.818241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:49.821126Z","iopub.execute_input":"2024-02-13T02:35:49.821515Z","iopub.status.idle":"2024-02-13T02:35:49.826524Z","shell.execute_reply.started":"2024-02-13T02:35:49.821474Z","shell.execute_reply":"2024-02-13T02:35:49.825578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = \"denselight_model\"\nmodel_3 = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/denselight_model.pkl\")\nmodel_3","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:49.827833Z","iopub.execute_input":"2024-02-13T02:35:49.828183Z","iopub.status.idle":"2024-02-13T02:35:50.289737Z","shell.execute_reply.started":"2024-02-13T02:35:49.828149Z","shell.execute_reply":"2024-02-13T02:35:50.288858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:50.290857Z","iopub.execute_input":"2024-02-13T02:35:50.291172Z","iopub.status.idle":"2024-02-13T02:35:50.298287Z","shell.execute_reply.started":"2024-02-13T02:35:50.291148Z","shell.execute_reply":"2024-02-13T02:35:50.297343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_3 = pd.Series(\n    predict_proba_in_batches(model_3, X_test, predict_mode = \"lightautoml\"),\n    index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_3","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:50.299398Z","iopub.execute_input":"2024-02-13T02:35:50.299738Z","iopub.status.idle":"2024-02-13T02:35:54.672445Z","shell.execute_reply.started":"2024-02-13T02:35:50.299715Z","shell.execute_reply":"2024-02-13T02:35:54.671588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/denselight_oof_preds.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:54.673580Z","iopub.execute_input":"2024-02-13T02:35:54.673844Z","iopub.status.idle":"2024-02-13T02:35:54.684290Z","shell.execute_reply.started":"2024-02-13T02:35:54.673821Z","shell.execute_reply":"2024-02-13T02:35:54.683379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:54.685607Z","iopub.execute_input":"2024-02-13T02:35:54.686198Z","iopub.status.idle":"2024-02-13T02:35:55.438431Z","shell.execute_reply.started":"2024-02-13T02:35:54.686164Z","shell.execute_reply":"2024-02-13T02:35:55.437535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Estimation Results","metadata":{}},{"cell_type":"code","source":"models_score_df","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:55.439724Z","iopub.execute_input":"2024-02-13T02:35:55.440099Z","iopub.status.idle":"2024-02-13T02:35:55.448886Z","shell.execute_reply.started":"2024-02-13T02:35:55.440051Z","shell.execute_reply":"2024-02-13T02:35:55.447900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:55.450122Z","iopub.execute_input":"2024-02-13T02:35:55.450752Z","iopub.status.idle":"2024-02-13T02:35:55.468971Z","shell.execute_reply.started":"2024-02-13T02:35:55.450715Z","shell.execute_reply":"2024-02-13T02:35:55.468155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"week_num = list(test_df[\"WEEK_NUM\"])\ndel test_df\ngc.collect()","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-13T02:35:55.474508Z","iopub.execute_input":"2024-02-13T02:35:55.474784Z","iopub.status.idle":"2024-02-13T02:35:56.048143Z","shell.execute_reply.started":"2024-02-13T02:35:55.474761Z","shell.execute_reply":"2024-02-13T02:35:56.047129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Blending","metadata":{}},{"cell_type":"code","source":"def gini_wrapper(base_df):\n    base_df = base_df[[\"WEEK_NUM\", \"target\"]].copy()\n    def gini_wrapper_inner(target, scores):\n        base_df[\"score\"] = scores\n        gini_score = gini_stability(base_df, score_col=\"score\")\n        return 1 - gini_score\n    return gini_wrapper_inner","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:56.049554Z","iopub.execute_input":"2024-02-13T02:35:56.049929Z","iopub.status.idle":"2024-02-13T02:35:56.056335Z","shell.execute_reply.started":"2024-02-13T02:35:56.049894Z","shell.execute_reply":"2024-02-13T02:35:56.055455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hill climbing using minimize","metadata":{}},{"cell_type":"code","source":"class WeightsSearcher:\n    def __init__(self, loss_fn, bounds=[], mode=\"min\", method='SLSQP'):\n        self.loss_fn = loss_fn\n        self.bounds = bounds\n        self.mode = mode\n        self.method = method # Nelder-Mead - for not smooth functions\n        \n    def _objective_function_wrapper(self, pred_values, true_targets, obj_fn):\n        def objective_function(weights):\n            pred_weighted = (pred_values * weights).sum(axis=1)\n            score = obj_fn(true_targets, pred_weighted)\n            return score\n        return objective_function\n    \n    def find_weights(self, val_preds, true_targets):\n        len_models = len(self.bounds)\n        bounds = [0,1] * len_models if len(self.bounds) == 0 else self.bounds\n        initial_weights = np.ones(len_models) / len_models\n        objective_function = self._objective_function_wrapper(val_preds, true_targets, self.loss_fn)\n        result = minimize(\n            objective_function, \n            initial_weights, \n            bounds=bounds, \n            method=self.method,\n        )\n        optimized_weights = result.x\n        optimized_weights /= np.sum(optimized_weights)\n        return optimized_weights","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:56.057612Z","iopub.execute_input":"2024-02-13T02:35:56.057886Z","iopub.status.idle":"2024-02-13T02:35:56.067574Z","shell.execute_reply.started":"2024-02-13T02:35:56.057864Z","shell.execute_reply":"2024-02-13T02:35:56.066661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_names = models_score_df.index.to_list()\npred_cols = [f\"pred_{name}\" for name in model_names]\nmodel_names","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:56.068685Z","iopub.execute_input":"2024-02-13T02:35:56.068965Z","iopub.status.idle":"2024-02-13T02:35:56.080767Z","shell.execute_reply.started":"2024-02-13T02:35:56.068935Z","shell.execute_reply":"2024-02-13T02:35:56.079928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bounds = [(0, 1)] * len(pred_cols)\nroc_auc_fn = lambda y_true, y_pred: 1 - roc_auc_score(y_true, y_pred)\ngini_score_fn = gini_wrapper(oof_df)\nw_searcher = WeightsSearcher(gini_score_fn, bounds, method='Nelder-Mead') # log_loss, gini_stability roc_auc_fn\noptimized_weights = w_searcher.find_weights(\n    oof_df[pred_cols].to_numpy(), \n    y_train\n)\noptimized_weights_df = pd.DataFrame(zip(model_names, optimized_weights), columns=['model', 'weight'])\ndisplay(optimized_weights_df)\nprint(\"sum: \", np.sum(optimized_weights))","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:35:56.081797Z","iopub.execute_input":"2024-02-13T02:35:56.082150Z","iopub.status.idle":"2024-02-13T02:37:01.740138Z","shell.execute_reply.started":"2024-02-13T02:35:56.082118Z","shell.execute_reply":"2024-02-13T02:37:01.739235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_pred_optimized = (oof_df[pred_cols] * optimized_weights).sum(axis=1).to_numpy()\noof_df[\"pred_optimized\"] = oof_pred_optimized\nroc_auc_oof = roc_auc_score(y_train, oof_pred_optimized)\ngini_score = gini_stability(oof_df, score_col=\"pred_optimized\")\nprint(\"CV roc_auc_oof optimized:\\t\", roc_auc_oof)\nprint(\"CV gini_score:\\t\\t\\t\", gini_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:37:01.741576Z","iopub.execute_input":"2024-02-13T02:37:01.741966Z","iopub.status.idle":"2024-02-13T02:37:03.684702Z","shell.execute_reply.started":"2024-02-13T02:37:01.741922Z","shell.execute_reply":"2024-02-13T02:37:03.683630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_all = (optimized_weights * test_preds_df[pred_cols]).sum(axis=1).to_numpy()\ny_pred_all[:10]","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:37:03.686108Z","iopub.execute_input":"2024-02-13T02:37:03.686797Z","iopub.status.idle":"2024-02-13T02:37:03.696606Z","shell.execute_reply.started":"2024-02-13T02:37:03.686760Z","shell.execute_reply":"2024-02-13T02:37:03.695616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"subm_df = pd.read_csv(data_nb.CFG.root_dir / \"sample_submission.csv\")\nsubm_df = subm_df.set_index(\"case_id\")\nsubm_df[\"score\"] = y_pred_all\nsubm_df[\"WEEK_NUM\"] = week_num\ndisplay(subm_df.head())\nprint(\"Check null: \", subm_df[\"score\"].isnull().any())","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:37:03.697750Z","iopub.execute_input":"2024-02-13T02:37:03.698116Z","iopub.status.idle":"2024-02-13T02:37:03.718789Z","shell.execute_reply.started":"2024-02-13T02:37:03.698064Z","shell.execute_reply":"2024-02-13T02:37:03.717869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"condition = subm_df[\"WEEK_NUM\"] < (subm_df[\"WEEK_NUM\"].max() - subm_df[\"WEEK_NUM\"].min())/2 + subm_df[\"WEEK_NUM\"].min() \nsubm_df.loc[condition, 'score'] = (subm_df.loc[condition, 'score'] - 0.02).clip(0) \nsubm_df","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:37:03.719906Z","iopub.execute_input":"2024-02-13T02:37:03.720206Z","iopub.status.idle":"2024-02-13T02:37:03.735134Z","shell.execute_reply.started":"2024-02-13T02:37:03.720183Z","shell.execute_reply":"2024-02-13T02:37:03.734130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del subm_df[\"WEEK_NUM\"]\nsubm_df.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:37:03.736367Z","iopub.execute_input":"2024-02-13T02:37:03.736668Z","iopub.status.idle":"2024-02-13T02:37:03.744203Z","shell.execute_reply.started":"2024-02-13T02:37:03.736644Z","shell.execute_reply":"2024-02-13T02:37:03.743352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df","metadata":{"execution":{"iopub.status.busy":"2024-02-13T02:37:03.745431Z","iopub.execute_input":"2024-02-13T02:37:03.745799Z","iopub.status.idle":"2024-02-13T02:37:03.757111Z","shell.execute_reply.started":"2024-02-13T02:37:03.745767Z","shell.execute_reply":"2024-02-13T02:37:03.756020Z"},"trusted":true},"execution_count":null,"outputs":[]}]}