{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom numpy.typing import ArrayLike, NDArray\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.base import BaseEstimator\nfrom functools import partial\nimport optuna","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:40.630086Z","iopub.execute_input":"2024-11-06T23:15:40.630503Z","iopub.status.idle":"2024-11-06T23:15:42.676353Z","shell.execute_reply.started":"2024-11-06T23:15:40.630462Z","shell.execute_reply":"2024-11-06T23:15:42.675143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = \"/kaggle/input/child-mind-institute-problematic-internet-use/\"","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.678264Z","iopub.execute_input":"2024-11-06T23:15:42.678889Z","iopub.status.idle":"2024-11-06T23:15:42.684171Z","shell.execute_reply.started":"2024-11-06T23:15:42.678837Z","shell.execute_reply":"2024-11-06T23:15:42.682858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(INPUT_DIR + \"train.csv\")\ndf_test = pd.read_csv(INPUT_DIR + \"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.685688Z","iopub.execute_input":"2024-11-06T23:15:42.686183Z","iopub.status.idle":"2024-11-06T23:15:42.761164Z","shell.execute_reply.started":"2024-11-06T23:15:42.686130Z","shell.execute_reply":"2024-11-06T23:15:42.760166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.dropna(subset=[\"sii\"]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.763642Z","iopub.execute_input":"2024-11-06T23:15:42.764031Z","iopub.status.idle":"2024-11-06T23:15:42.774657Z","shell.execute_reply.started":"2024-11-06T23:15:42.763991Z","shell.execute_reply":"2024-11-06T23:15:42.773516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cols = list(set(df.columns) - set(df_test.columns))\n\nX = df.drop([\"id\"] + drop_cols, axis=1)\ny = df[\"sii\"].astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.776416Z","iopub.execute_input":"2024-11-06T23:15:42.776781Z","iopub.status.idle":"2024-11-06T23:15:42.788535Z","shell.execute_reply.started":"2024-11-06T23:15:42.776743Z","shell.execute_reply":"2024-11-06T23:15:42.787420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = X.columns[X.dtypes==\"object\"].tolist()\nif cat_cols:\n    X[cat_cols] = X[cat_cols].astype(\"category\")","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.790095Z","iopub.execute_input":"2024-11-06T23:15:42.790547Z","iopub.status.idle":"2024-11-06T23:15:42.814365Z","shell.execute_reply.started":"2024-11-06T23:15:42.790493Z","shell.execute_reply":"2024-11-06T23:15:42.813326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = y.mean()\nb = y.var(ddof=0)\n\ny_min = y.min()\ny_max = y.max()","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.815639Z","iopub.execute_input":"2024-11-06T23:15:42.816025Z","iopub.status.idle":"2024-11-06T23:15:42.822735Z","shell.execute_reply.started":"2024-11-06T23:15:42.815986Z","shell.execute_reply":"2024-11-06T23:15:42.821503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(preds, data):\n    y_true = data.get_label()\n    y_pred = preds.clip(y_min, y_max).round()\n    qwk = cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n    return 'QWK', qwk, True\n\n\ndef qwk_obj(preds, dtrain):\n    labels = dtrain.get_label()\n    preds = preds.clip(y_min, y_max)\n    f = 1/2 * np.sum((preds - labels)**2)\n    g = 1/2 * np.sum((preds - a)**2 + b)\n    df = preds - labels\n    dg = preds - a\n    grad = (df/g - f*dg/g**2)*len(labels)\n    hess = np.ones(len(labels))\n    return grad, hess","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.824384Z","iopub.execute_input":"2024-11-06T23:15:42.824762Z","iopub.status.idle":"2024-11-06T23:15:42.835325Z","shell.execute_reply.started":"2024-11-06T23:15:42.824708Z","shell.execute_reply":"2024-11-06T23:15:42.834112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    \"objective\": qwk_obj,\n    \"metric\": \"None\",\n    \"verbosity\": -1,\n    \"learning_rate\": 0.01,\n    \"num_leaves\": 16,\n    \"feature_fraction\": 0.5\n}\n\n# The initial score of the model is the key parameter.\n# I found that the mean value of the target is a bad choice for this data.\ninit_score = 2.0","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.836718Z","iopub.execute_input":"2024-11-06T23:15:42.837281Z","iopub.status.idle":"2024-11-06T23:15:42.846906Z","shell.execute_reply.started":"2024-11-06T23:15:42.837227Z","shell.execute_reply":"2024-11-06T23:15:42.845735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=0)\nfolds = [(idx_train, idx_valid) for idx_train, idx_valid in skf.split(X, y)]\n\nmodels = lgb.cv(\n    params=params,\n    train_set=lgb.Dataset(X, y, init_score=[init_score]*len(X)),\n    num_boost_round=10000,\n    folds=folds,\n    feval=quadratic_weighted_kappa,\n    callbacks=[\n        lgb.early_stopping(stopping_rounds=100, verbose=True),\n        lgb.log_evaluation(100)\n    ],\n    return_cvbooster=True\n)[\"cvbooster\"].boosters\n\npreds_oof = np.zeros(len(df))\nfor model, (idx_train, idx_valid) in zip(models, folds):\n    preds_oof[idx_valid] = model.predict(X.iloc[idx_valid]) + init_score","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:15:42.851831Z","iopub.execute_input":"2024-11-06T23:15:42.852875Z","iopub.status.idle":"2024-11-06T23:16:08.662876Z","shell.execute_reply.started":"2024-11-06T23:15:42.852743Z","shell.execute_reply":"2024-11-06T23:16:08.661639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OptimizedRounder:\n    \"\"\"\n    A class for optimizing the rounding of continuous predictions into discrete class labels using Optuna.\n    The optimization process maximizes the Quadratic Weighted Kappa score by learning thresholds that separate\n    continuous predictions into class intervals.\n\n    Args:\n        n_classes (int): The number of discrete class labels.\n        n_trials (int, optional): The number of trials for the Optuna optimization. Defaults to 100.\n\n    Attributes:\n        n_classes (int): The number of discrete class labels.\n        labels (NDArray[np.int_]): An array of class labels from 0 to `n_classes - 1`.\n        n_trials (int): The number of optimization trials.\n        metric (Callable): The Quadratic Weighted Kappa score metric used for optimization.\n        thresholds (List[float]): The optimized thresholds learned after calling `fit()`.\n\n    Methods:\n        fit(y_pred: NDArray[np.float_], y_true: NDArray[np.int_]) -> None:\n            Fits the rounding thresholds based on continuous predictions and ground truth labels.\n\n            Args:\n                y_pred (NDArray[np.float_]): Continuous predictions that need to be rounded.\n                y_true (NDArray[np.int_]): Ground truth class labels.\n\n            Returns:\n                None\n\n        predict(y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n            Predicts discrete class labels by rounding continuous predictions using the fitted thresholds.\n            `fit()` must be called before `predict()`.\n\n            Args:\n                y_pred (NDArray[np.float_]): Continuous predictions to be rounded.\n\n            Returns:\n                NDArray[np.int_]: Predicted class labels.\n\n        _normalize(y: NDArray[np.float_]) -> NDArray[np.float_]:\n            Normalizes the continuous values to the range [0, `n_classes - 1`].\n\n            Args:\n                y (NDArray[np.float_]): Continuous values to be normalized.\n\n            Returns:\n                NDArray[np.float_]: Normalized values.\n\n    References:\n        - This implementation uses Optuna for threshold optimization.\n        - Quadratic Weighted Kappa is used as the evaluation metric.\n    \"\"\"\n\n    def __init__(self, n_classes: int, n_trials: int = 100):\n        self.n_classes = n_classes\n        self.labels = np.arange(n_classes)\n        self.n_trials = n_trials\n        self.metric = partial(cohen_kappa_score, weights=\"quadratic\")\n\n    def fit(self, y_pred: NDArray[np.float_], y_true: NDArray[np.int_]) -> None:\n        y_pred = self._normalize(y_pred)\n\n        def objective(trial: optuna.Trial) -> float:\n            thresholds = []\n            for i in range(self.n_classes - 1):\n                low = max(thresholds) if i > 0 else min(self.labels)\n                high = max(self.labels)\n                th = trial.suggest_float(f\"threshold_{i}\", low, high)\n                thresholds.append(th)\n            try:\n                y_pred_rounded = np.digitize(y_pred, thresholds)\n            except ValueError:\n                return -100\n            return self.metric(y_true, y_pred_rounded)\n\n        optuna.logging.disable_default_handler()\n        study = optuna.create_study(direction=\"maximize\")\n        study.optimize(\n            objective,\n            n_trials=self.n_trials,\n        )\n        self.thresholds = [study.best_params[f\"threshold_{i}\"] for i in range(self.n_classes - 1)]\n\n    def predict(self, y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n        assert hasattr(self, \"thresholds\"), \"fit() must be called before predict()\"\n        y_pred = self._normalize(y_pred)\n        return np.digitize(y_pred, self.thresholds)\n\n    def _normalize(self, y: NDArray[np.float_]) -> NDArray[np.float_]:\n        # normalize y_pred to [0, n_classes - 1]\n        return (y - y.min()) / (y.max() - y.min()) * (self.n_classes - 1)","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:16:08.665271Z","iopub.execute_input":"2024-11-06T23:16:08.665884Z","iopub.status.idle":"2024-11-06T23:16:08.684388Z","shell.execute_reply.started":"2024-11-06T23:16:08.665829Z","shell.execute_reply":"2024-11-06T23:16:08.683225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# without threshold optimization\n# preds_oof = preds_oof.clip(y_min, y_max).round()","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:16:08.685980Z","iopub.execute_input":"2024-11-06T23:16:08.686474Z","iopub.status.idle":"2024-11-06T23:16:08.702427Z","shell.execute_reply.started":"2024-11-06T23:16:08.686420Z","shell.execute_reply":"2024-11-06T23:16:08.701160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = OptimizedRounder(n_classes=4, n_trials=300)\noptimizer.fit(preds_oof, y)\npreds_oof_rounded = optimizer.predict(preds_oof)","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:16:08.704226Z","iopub.execute_input":"2024-11-06T23:16:08.704696Z","iopub.status.idle":"2024-11-06T23:16:16.681007Z","shell.execute_reply.started":"2024-11-06T23:16:08.704642Z","shell.execute_reply":"2024-11-06T23:16:16.679775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qwk = cohen_kappa_score(y, preds_oof_rounded, weights=\"quadratic\")\nprint(\"QWK:\", qwk)","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:16:16.683599Z","iopub.execute_input":"2024-11-06T23:16:16.683985Z","iopub.status.idle":"2024-11-06T23:16:16.693097Z","shell.execute_reply.started":"2024-11-06T23:16:16.683946Z","shell.execute_reply":"2024-11-06T23:16:16.691855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AvgModel:\n    def __init__(self, models: list[BaseEstimator]):\n        self.models = models\n\n    def predict(self, X: ArrayLike) -> NDArray[np.int_]:\n        preds: list[NDArray[np.int_]] = []\n        for model in self.models:\n            pred = model.predict(X)\n            preds.append(pred)\n\n        return np.mean(preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:16:16.694536Z","iopub.execute_input":"2024-11-06T23:16:16.694943Z","iopub.status.idle":"2024-11-06T23:16:16.708075Z","shell.execute_reply.started":"2024-11-06T23:16:16.694904Z","shell.execute_reply":"2024-11-06T23:16:16.706955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop([\"id\"], axis=1)\n\ncat_cols = X_test.columns[X_test.dtypes==\"object\"].tolist()\nif cat_cols:\n    X_test[cat_cols] = X_test[cat_cols].astype(\"category\")","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:16:16.709453Z","iopub.execute_input":"2024-11-06T23:16:16.709786Z","iopub.status.idle":"2024-11-06T23:16:16.732602Z","shell.execute_reply.started":"2024-11-06T23:16:16.709749Z","shell.execute_reply":"2024-11-06T23:16:16.731261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_model = AvgModel(models)\ntest_pred = avg_model.predict(X_test) + init_score\ntest_pred_rounded = optimizer.predict(test_pred)","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:16:16.734244Z","iopub.execute_input":"2024-11-06T23:16:16.735067Z","iopub.status.idle":"2024-11-06T23:16:16.789544Z","shell.execute_reply.started":"2024-11-06T23:16:16.735009Z","shell.execute_reply":"2024-11-06T23:16:16.788182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = df_test[[\"id\"]].assign(sii=test_pred_rounded)\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-06T23:16:16.791255Z","iopub.execute_input":"2024-11-06T23:16:16.791754Z","iopub.status.idle":"2024-11-06T23:16:16.802421Z","shell.execute_reply.started":"2024-11-06T23:16:16.791694Z","shell.execute_reply":"2024-11-06T23:16:16.801108Z"},"trusted":true},"execution_count":null,"outputs":[]}]}