{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":8411082,"sourceType":"datasetVersion","datasetId":4550700}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### Implement Home Credit competition's stability metric into Optuna objective","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport optuna\nfrom sklearn.metrics import roc_auc_score\nfrom lightgbm import LGBMClassifier, early_stopping","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-21T09:35:08.186723Z","iopub.execute_input":"2025-11-21T09:35:08.187007Z","iopub.status.idle":"2025-11-21T09:35:12.29461Z","shell.execute_reply.started":"2025-11-21T09:35:08.18698Z","shell.execute_reply":"2025-11-21T09:35:12.293347Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We suppose that we have already all the necessary variables (X_train, X_val and other further used)","metadata":{}},{"cell_type":"code","source":"class StabilityMetric:\n    \"\"\"\n    Stability metric for model optimization during training\n    \"\"\"\n    def __init__(self, week_num, X_val):\n        self.X_val = X_val\n        self.week_num = week_num       \n\n    def metric_func(self, y_true, y_pred):\n        gini_in_time = []\n        weeks_to_score = self.week_num[self.X_val.index].reset_index(drop=True)\n        \n        for week in weeks_to_score.unique():\n            week_idx = weeks_to_score.eq(week)\n            gini = 2 * roc_auc_score(y_true[week_idx], y_pred[week_idx]) - 1\n            gini_in_time.append(gini)\n            \n        w_fallingrate = 88.0\n        w_resstd = -0.5\n        x = np.arange(len(gini_in_time))\n        y = np.array(gini_in_time)\n        a, b = np.polyfit(x, y, 1)\n        y_hat = a * x + b\n        residuals = y - y_hat\n        res_std = np.std(residuals)\n        avg_gini = np.mean(y)\n        stability_score = avg_gini + w_fallingrate * min(0, a) + w_resstd * res_std\n        \n        is_higher_better = True\n        \n        return 'stability_score', stability_score, is_higher_better\n\ndef model_objective(trial, X_train, y_train, X_val, y_val, week_num, device='gpu'):\n    \"\"\"\n    Objective function for hyperparameter tuning\n    \"\"\" \n    # Target ratio\n    y_ratio = np.sum(y_train == 0) / np.sum(y_train == 1)\n    \n    params = {      \n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),\n            'num_leaves': trial.suggest_int('num_leaves', 8, 256),\n            'min_child_samples': trial.suggest_int('min_child_samples', 5, 200),\n            'reg_alpha': trial.suggest_float('reg_alpha', 1e-8, 10.0, log=True),\n            'reg_lambda': trial.suggest_float('reg_lambda', 1e-8, 10.0, log=True),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.4, 1.0),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'subsample_freq': trial.suggest_int('subsample_freq', 0, 10),\n            'scale_pos_weight': trial.suggest_float('scale_pos_weight', 1.0, y_ratio*2, log=True),\n\n            'n_estimators': 3000,\n            'objective': 'binary',\n            'metric': 'None',\n            'verbosity': -1,\n            'device': device,\n            'max_bin': 255,\n            'n_jobs': -1,\n            }\n    stability_metric = StabilityMetric(week_num, X_val)   \n    model = LGBMClassifier(**params)  \n    model.fit(X_train, y_train, \n              eval_set=[(X_val, y_val)],\n              eval_metric=stability_metric.metric_func,\n              callbacks=[early_stopping(50)],\n             ) \n    y_pred = model.predict_proba(X_val)[:, 1]\n    _, stability_score, _ = stability_metric.metric_func(\n        np.array(y_val), np.array(y_pred)\n    )\n    return stability_score\n\ndef objective(trial):\n    return model_objective(trial, X_train, y_train, X_val, y_val, \n                           week_num, device)\n# Optuna study\noptuna_test = False\n\nif optuna_test:\n    sampler = optuna.samplers.TPESampler(multivariate=True, seed=42)\n    study = optuna.create_study(direction='maximize', sampler=sampler)\n    study.optimize(objective, n_trials=50, timeout=60*60*12)  # 12 hours timeout\n    \n    # Show best results\n    trial = study.best_trial\n    completed_trials = [t for t in study.trials if t.state == optuna.trial.TrialState.COMPLETE]\n    print(f'Number of completed trials: {len(completed_trials)}')\n    print(f'Best score: {trial.value:.3f}')\n    print(f'Best params: {trial.params}')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-28T14:05:17.555163Z","iopub.execute_input":"2025-07-28T14:05:17.55554Z","iopub.status.idle":"2025-07-28T14:05:17.571506Z","shell.execute_reply.started":"2025-07-28T14:05:17.55551Z","shell.execute_reply":"2025-07-28T14:05:17.569618Z"}},"outputs":[],"execution_count":null}]}