{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"},{"sourceId":162438049,"sourceType":"kernelVersion"},{"sourceId":162470947,"sourceType":"kernelVersion"},{"sourceId":162602153,"sourceType":"kernelVersion"},{"sourceId":162605422,"sourceType":"kernelVersion"},{"sourceId":162611024,"sourceType":"kernelVersion"}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dependencies","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies pandas==2.0.3","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-02-15T12:30:49.406802Z","iopub.execute_input":"2024-02-15T12:30:49.407174Z","iopub.status.idle":"2024-02-15T12:31:14.908958Z","shell.execute_reply.started":"2024-02-15T12:30:49.407135Z","shell.execute_reply":"2024-02-15T12:31:14.907948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nfrom glob import glob\nfrom pathlib import Path\nfrom datetime import datetime\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom sklearn.metrics import roc_auc_score, log_loss\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.preprocessing import LabelEncoder\nfrom scipy.optimize import minimize\n\nimport joblib\nimport lightgbm as lgb\nimport warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-15T12:31:14.910952Z","iopub.execute_input":"2024-02-15T12:31:14.911263Z","iopub.status.idle":"2024-02-15T12:31:19.702861Z","shell.execute_reply.started":"2024-02-15T12:31:14.911234Z","shell.execute_reply":"2024-02-15T12:31:19.702028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"code","source":"import home_credit_baseline_data as data_nb","metadata":{"execution":{"iopub.status.busy":"2024-02-15T12:31:19.704147Z","iopub.execute_input":"2024-02-15T12:31:19.704520Z","iopub.status.idle":"2024-02-15T12:31:19.804923Z","shell.execute_reply.started":"2024-02-15T12:31:19.704486Z","shell.execute_reply":"2024-02-15T12:31:19.803945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### from https://www.kaggle.com/code/batprem/home-credit-risk-mode-utility-scripts\n\ndef gini_stability(base, score_col=\"score\", w_fallingrate=88.0, w_resstd=-0.5):\n    gini_in_time = base.loc[:, [\"WEEK_NUM\", \"target\", score_col]]\\\n        .sort_values(\"WEEK_NUM\")\\\n        .groupby(\"WEEK_NUM\")[[\"target\", score_col]]\\\n        .apply(lambda x: 2*roc_auc_score(x[\"target\"], x[score_col])-1).tolist()\n    \n    x = np.arange(len(gini_in_time))\n    y = gini_in_time\n    a, b = np.polyfit(x, y, 1)\n    y_hat = a*x + b\n    residuals = y - y_hat\n    res_std = np.std(residuals)\n    avg_gini = np.mean(gini_in_time)\n    return avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std","metadata":{"execution":{"iopub.status.busy":"2024-02-15T12:31:19.807133Z","iopub.execute_input":"2024-02-15T12:31:19.807629Z","iopub.status.idle":"2024-02-15T12:31:19.815148Z","shell.execute_reply.started":"2024-02-15T12:31:19.807601Z","shell.execute_reply":"2024-02-15T12:31:19.814295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_proba_in_batches(model, data, batch_size=100000, predict_mode=\"base\"):\n    num_samples = len(data)\n    num_batches = int(np.ceil(num_samples / batch_size))\n    probabilities = np.zeros((num_samples,))\n\n    for batch_idx in range(num_batches):\n        print(f\"Processing batch: {batch_idx+1}/{num_batches}\")\n        start_idx = batch_idx * batch_size\n        end_idx = min((batch_idx + 1) * batch_size, num_samples)\n        X_batch = data.iloc[start_idx:end_idx]\n        if predict_mode == \"base\":\n            batch_probs = model.predict_proba(X_batch)[:, 1]\n        elif predict_mode == \"lightautoml\":\n            batch_probs = model.predict(X_batch).data.squeeze()\n        probabilities[start_idx:end_idx] = batch_probs\n        gc.collect()\n\n    return probabilities","metadata":{"execution":{"iopub.status.busy":"2024-02-15T12:31:19.816608Z","iopub.execute_input":"2024-02-15T12:31:19.817110Z","iopub.status.idle":"2024-02-15T12:31:19.827450Z","shell.execute_reply.started":"2024-02-15T12:31:19.817078Z","shell.execute_reply":"2024-02-15T12:31:19.826636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_df = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\")\ny_train = train_base_df[\"target\"]\noof_df = train_base_df\nmodels_score_df = pd.DataFrame()\n# model_names = []\ntest_preds_df = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2024-02-15T12:31:19.828515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 1","metadata":{}},{"cell_type":"code","source":"model_name = \"model_1\"\nmodel_1 = joblib.load(\"/kaggle/input/home-credit-baseline-training/oof_model_1.pkl\")\nmodel_1\n# model_names.append(model_name)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = data_nb.prepare_df(data_nb.CFG.test_dir, cat_cols=cat_cols, mode=\"test\", train_cols=train_cols)\ndisplay(test_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds_df['case_id'] = test_df['case_id']\ntest_preds_df.set_index('case_id', inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_1 = pd.Series(predict_proba_in_batches(model_1, X_test), index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training/oof_pred.pkl\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 2","metadata":{}},{"cell_type":"code","source":"model_name = \"model_2\"\nmodel_2 = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/oof_model.pkl\")\nmodel_2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_2 = pd.Series(predict_proba_in_batches(model_2, X_test), index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training-model-2/oof_pred.pkl\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 3","metadata":{}},{"cell_type":"code","source":"!pip install --no-index -Uq --find-links=/kaggle/input/lightautoml-038-dependencies lightautoml==0.3.8","metadata":{"_kg_hide-output":true,"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = \"denselight_model\"\nmodel_3 = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/denselight_model.pkl\")\nmodel_3","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols, cat_cols, drop_cols = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/train_cat_columns.pkl\")\nprint(\"train_cols:\\t\", len(train_cols))\nprint(\"cat_cols:\\t\", len(cat_cols))\nprint(\"drop_cols:\\t\", len(drop_cols))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df.drop(columns=[\"WEEK_NUM\"] + drop_cols)\nX_test = X_test.set_index(\"case_id\")\nprint(\"X_test shape: \", X_test.shape)\n\ny_pred_3 = pd.Series(\n    predict_proba_in_batches(model_3, X_test, predict_mode = \"lightautoml\"),\n    index=X_test.index)\ntest_preds_df[f\"pred_{model_name}\"] = y_pred_3","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df[f\"pred_{model_name}\"] = joblib.load(\"/kaggle/input/home-credit-baseline-training-lightautoml/denselight_oof_preds.pkl\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gini_score = gini_stability(oof_df, score_col=f\"pred_{model_name}\")\nmodels_score_df.loc[model_name, [\"gini_score\"]] = gini_score\nprint(\"gini_score:\\t\", gini_score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Estimation Results","metadata":{}},{"cell_type":"code","source":"models_score_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"week_num = list(test_df[\"WEEK_NUM\"])\ndel test_df\ngc.collect()","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Blending","metadata":{}},{"cell_type":"code","source":"def gini_wrapper(base_df):\n    base_df = base_df[[\"WEEK_NUM\", \"target\"]].copy()\n    def gini_wrapper_inner(target, scores):\n        base_df[\"score\"] = scores\n        gini_score = gini_stability(base_df, score_col=\"score\")\n        return 1 - gini_score\n    return gini_wrapper_inner","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hill climbing using minimize","metadata":{}},{"cell_type":"code","source":"class WeightsSearcher:\n    def __init__(self, loss_fn, bounds=[], mode=\"min\", method='SLSQP'):\n        self.loss_fn = loss_fn\n        self.bounds = bounds\n        self.mode = mode\n        self.method = method # Nelder-Mead - for not smooth functions\n        \n    def _objective_function_wrapper(self, pred_values, true_targets, obj_fn):\n        def objective_function(weights):\n            pred_weighted = (pred_values * weights).sum(axis=1)\n            score = obj_fn(true_targets, pred_weighted)\n            return score\n        return objective_function\n    \n    def find_weights(self, val_preds, true_targets):\n        len_models = len(self.bounds)\n        bounds = [0,1] * len_models if len(self.bounds) == 0 else self.bounds\n        initial_weights = np.ones(len_models) / len_models\n        objective_function = self._objective_function_wrapper(val_preds, true_targets, self.loss_fn)\n        result = minimize(\n            objective_function, \n            initial_weights, \n            bounds=bounds, \n            method=self.method,\n        )\n        optimized_weights = result.x\n        optimized_weights /= np.sum(optimized_weights)\n        return optimized_weights","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_names = models_score_df.index.to_list()\npred_cols = [f\"pred_{name}\" for name in model_names]\nmodel_names","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bounds = [(0, 1)] * len(pred_cols)\nroc_auc_fn = lambda y_true, y_pred: 1 - roc_auc_score(y_true, y_pred)\ngini_score_fn = gini_wrapper(oof_df)\nw_searcher = WeightsSearcher(gini_score_fn, bounds, method='Nelder-Mead') # log_loss, gini_stability roc_auc_fn\noptimized_weights = w_searcher.find_weights(\n    oof_df[pred_cols].to_numpy(), \n    y_train\n)\noptimized_weights_df = pd.DataFrame(zip(model_names, optimized_weights), columns=['model', 'weight'])\ndisplay(optimized_weights_df)\nprint(\"sum: \", np.sum(optimized_weights))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_pred_optimized = (oof_df[pred_cols] * optimized_weights).sum(axis=1).to_numpy()\noof_df[\"pred_optimized\"] = oof_pred_optimized\nroc_auc_oof = roc_auc_score(y_train, oof_pred_optimized)\ngini_score = gini_stability(oof_df, score_col=\"pred_optimized\")\nprint(\"CV roc_auc_oof optimized:\\t\", roc_auc_oof)\nprint(\"CV gini_score:\\t\\t\\t\", gini_score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_all = (optimized_weights * test_preds_df[pred_cols]).sum(axis=1).to_numpy()\ny_pred_all[:10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"subm_df = pd.read_csv(data_nb.CFG.root_dir / \"sample_submission.csv\")\nsubm_df = subm_df.set_index(\"case_id\")\nsubm_df[\"score\"] = y_pred_all\nsubm_df[\"WEEK_NUM\"] = week_num\ndisplay(subm_df.head())\nprint(\"Check null: \", subm_df[\"score\"].isnull().any())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"condition = subm_df[\"WEEK_NUM\"] < (subm_df[\"WEEK_NUM\"].max() - subm_df[\"WEEK_NUM\"].min())/2 + subm_df[\"WEEK_NUM\"].min() \nsubm_df.loc[condition, 'score'] = (subm_df.loc[condition, 'score'] - 0.02).clip(0) \nsubm_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del subm_df[\"WEEK_NUM\"]\nsubm_df.to_csv(\"submission.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}