{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":197516177,"sourceType":"kernelVersion"},{"sourceId":197506774,"sourceType":"kernelVersion"}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"LAUNCH_VARIANT,ENSEMBLE_SOLUTIONS = 'option 52',['SOLUTION_1']","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:16:37.504647Z","iopub.execute_input":"2024-09-28T12:16:37.504944Z","iopub.status.idle":"2024-09-28T12:16:37.515045Z","shell.execute_reply.started":"2024-09-28T12:16:37.504907Z","shell.execute_reply":"2024-09-28T12:16:37.514094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    !nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:16:37.516574Z","iopub.execute_input":"2024-09-28T12:16:37.516867Z","iopub.status.idle":"2024-09-28T12:16:38.602710Z","shell.execute_reply.started":"2024-09-28T12:16:37.516835Z","shell.execute_reply":"2024-09-28T12:16:38.601614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    !pip install /kaggle/input/polars-gpu-1-7-1/cupy_cuda12x-13.3.0-cp310-cp310-manylinux2014_x86_64.whl\n    !pip install /kaggle/input/polars-gpu-1-7-1/rmm_cu12-24.8.2-cp310-cp310-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl\n    !pip install /kaggle/input/polars-gpu-1-7-1/cudf_cu12-24.8.3-cp310-cp310-manylinux_2_28_x86_64.whl\n    !pip install /kaggle/input/polars-gpu-1-7-1/polars-1.7.1-cp38-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n    !pip install /kaggle/input/polars-gpu-1-7-1/cudf_polars_cu12-24.8.3-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:16:38.604368Z","iopub.execute_input":"2024-09-28T12:16:38.604804Z","iopub.status.idle":"2024-09-28T12:19:33.791575Z","shell.execute_reply.started":"2024-09-28T12:16:38.604767Z","shell.execute_reply":"2024-09-28T12:19:33.790437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n\n    import warnings\n    from functools import partial\n    from pathlib import Path\n\n    from tqdm import tqdm\n    import matplotlib.pyplot as plt\n    import numpy as np\n    import optuna\n    import polars as pl\n    import polars.selectors as cs\n    from catboost import CatBoostRegressor, MultiTargetCustomMetric\n    from numpy.typing import ArrayLike, NDArray\n    from polars.testing import assert_frame_equal\n    from sklearn.base import BaseEstimator\n    from sklearn.metrics import cohen_kappa_score\n    from sklearn.model_selection import StratifiedKFold\n\n    warnings.filterwarnings(\"ignore\", message=\"Failed to optimize method\")\n\n    DATA_DIR = Path(\"/kaggle/input/child-mind-institute-problematic-internet-use\")\n    TARGET_COLS = [\n        \"PCIAT-PCIAT_01\",\n        \"PCIAT-PCIAT_02\",\n        \"PCIAT-PCIAT_03\",\n        \"PCIAT-PCIAT_04\",\n        \"PCIAT-PCIAT_05\",\n        \"PCIAT-PCIAT_06\",\n        \"PCIAT-PCIAT_07\",\n        \"PCIAT-PCIAT_08\",\n        \"PCIAT-PCIAT_09\",\n        \"PCIAT-PCIAT_10\",\n        \"PCIAT-PCIAT_11\",\n        \"PCIAT-PCIAT_12\",\n        \"PCIAT-PCIAT_13\",\n        \"PCIAT-PCIAT_14\",\n        \"PCIAT-PCIAT_15\",\n        \"PCIAT-PCIAT_16\",\n        \"PCIAT-PCIAT_17\",\n        \"PCIAT-PCIAT_18\",\n        \"PCIAT-PCIAT_19\",\n        \"PCIAT-PCIAT_20\",\n        \"PCIAT-PCIAT_Total\",\n        \"sii\",\n    ]\n\n    FEATURE_COLS = [\n        \"Basic_Demos-Enroll_Season\",\n        \"Basic_Demos-Age\",\n        \"Basic_Demos-Sex\",\n        \"CGAS-Season\",\n        \"CGAS-CGAS_Score\",\n        \"Physical-Season\",\n        \"Physical-BMI\",\n        \"Physical-Height\",\n        \"Physical-Weight\",\n        \"Physical-Waist_Circumference\",\n        \"Physical-Diastolic_BP\",\n        \"Physical-HeartRate\",\n        \"Physical-Systolic_BP\",\n        \"Fitness_Endurance-Season\",\n        \"Fitness_Endurance-Max_Stage\",\n        \"Fitness_Endurance-Time_Mins\",\n        \"Fitness_Endurance-Time_Sec\",\n        \"FGC-Season\",\n        \"FGC-FGC_CU\",\n        \"FGC-FGC_CU_Zone\",\n        \"FGC-FGC_GSND\",\n        \"FGC-FGC_GSND_Zone\",\n        \"FGC-FGC_GSD\",\n        \"FGC-FGC_GSD_Zone\",\n        \"FGC-FGC_PU\",\n        \"FGC-FGC_PU_Zone\",\n        \"FGC-FGC_SRL\",\n        \"FGC-FGC_SRL_Zone\",\n        \"FGC-FGC_SRR\",\n        \"FGC-FGC_SRR_Zone\",\n        \"FGC-FGC_TL\",\n        \"FGC-FGC_TL_Zone\",\n        \"BIA-Season\",\n        \"BIA-BIA_Activity_Level_num\",\n        \"BIA-BIA_BMC\",\n        \"BIA-BIA_BMI\",\n        \"BIA-BIA_BMR\",\n        \"BIA-BIA_DEE\",\n        \"BIA-BIA_ECW\",\n        \"BIA-BIA_FFM\",\n        \"BIA-BIA_FFMI\",\n        \"BIA-BIA_FMI\",\n        \"BIA-BIA_Fat\",\n        \"BIA-BIA_Frame_num\",\n        \"BIA-BIA_ICW\",\n        \"BIA-BIA_LDM\",\n        \"BIA-BIA_LST\",\n        \"BIA-BIA_SMM\",\n        \"BIA-BIA_TBW\",\n        \"PAQ_A-Season\",\n        \"PAQ_A-PAQ_A_Total\",\n        \"PAQ_C-Season\",\n        \"PAQ_C-PAQ_C_Total\",\n        \"SDS-Season\",\n        \"SDS-SDS_Total_Raw\",\n        \"SDS-SDS_Total_T\",\n        \"PreInt_EduHx-Season\",\n        \"PreInt_EduHx-computerinternet_hoursday\",\n\n        # stats features from parquets\n        \"X_min\",\n        \"Y_min\",\n        \"Z_min\",\n        \"enmo_min\",\n        \"anglez_min\",\n        \"light_min\",\n        \"battery_voltage_min\",\n        \"X_mean\",\n        \"Y_mean\",\n        \"Z_mean\",\n        \"enmo_mean\",\n        \"anglez_mean\",\n        \"light_mean\",\n        \"battery_voltage_mean\",\n        \"X_max\",\n        \"Y_max\",\n        \"Z_max\",\n        \"enmo_max\",\n        \"anglez_max\",\n        \"light_max\",\n        \"battery_voltage_max\",\n        \"X_std\",\n        \"Y_std\",\n        \"Z_std\",\n        \"enmo_std\",\n        \"anglez_std\",\n        \"light_std\",\n        \"battery_voltage_std\",\n    ]","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:19:33.794692Z","iopub.execute_input":"2024-09-28T12:19:33.795283Z","iopub.status.idle":"2024-09-28T12:19:35.555035Z","shell.execute_reply.started":"2024-09-28T12:19:33.795233Z","shell.execute_reply":"2024-09-28T12:19:35.554244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:    \n    \n    # Load data\n    train = pl.read_csv(DATA_DIR / \"train.csv\")\n    test = pl.read_csv(DATA_DIR / \"test.csv\")\n    train_test = pl.concat([train, test], how=\"diagonal\")\n\n    IS_TEST = test.height <= 100\n\n    assert_frame_equal(train, train_test[: train.height].select(train.columns))\n    assert_frame_equal(test, train_test[train.height :].select(test.columns))","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:19:35.556154Z","iopub.execute_input":"2024-09-28T12:19:35.556631Z","iopub.status.idle":"2024-09-28T12:19:35.694459Z","shell.execute_reply.started":"2024-09-28T12:19:35.556595Z","shell.execute_reply":"2024-09-28T12:19:35.693447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    # Cast string columns to categorical\n    train_test = train_test.with_columns(cs.string().cast(pl.Categorical).fill_null(\"NAN\"))\n    train = train_test[: train.height]\n    test = train_test[train.height :]","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:19:35.698041Z","iopub.execute_input":"2024-09-28T12:19:35.698705Z","iopub.status.idle":"2024-09-28T12:19:35.730228Z","shell.execute_reply.started":"2024-09-28T12:19:35.698649Z","shell.execute_reply":"2024-09-28T12:19:35.729260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    def split_array(ar, n_group):\n        for i_chunk in range(n_group):\n            yield ar[i_chunk * len(ar) // n_group : (i_chunk + 1) * len(ar) // n_group]\n\n    def agg_parquets(files):\n        cols = [\"X\", \"Y\", \"Z\", \"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n        aggs = []\n        files_chunks = list(split_array(files, 10))\n        for files_tmp in tqdm(files_chunks):\n            if len(files_tmp) == 0:\n                continue\n            dfs = []\n            for file in files_tmp:\n                df = pl.scan_parquet(file)\n                df = df.with_columns(pl.lit(file.parts[-1].split(\"=\")[1]).alias(\"id\"))\n                dfs.append(df)\n            df = pl.concat(dfs)\n            agg = (\n                df.group_by(\"id\")\n                .agg(\n                    [pl.col(c).cast(pl.Float32).min().alias(f\"{c}_min\") for c in cols]\n                    + [pl.col(c).cast(pl.Float32).mean().alias(f\"{c}_mean\") for c in cols]\n                    + [pl.col(c).cast(pl.Float32).max().alias(f\"{c}_max\") for c in cols]\n                    + [pl.col(c).cast(pl.Float32).std().alias(f\"{c}_std\") for c in cols]\n                )\n                .collect(engine=\"gpu\")\n            )\n            aggs.append(agg)\n        return pl.concat(aggs)\n\n    train_agg = agg_parquets(sorted(Path(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\").glob(\"*\")))\n    test_agg = agg_parquets(sorted(Path(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\").glob(\"*\")))","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:19:35.731631Z","iopub.execute_input":"2024-09-28T12:19:35.732302Z","iopub.status.idle":"2024-09-28T12:21:04.208546Z","shell.execute_reply.started":"2024-09-28T12:19:35.732265Z","shell.execute_reply":"2024-09-28T12:21:04.207488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:    \n    \n    train_agg","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.210020Z","iopub.execute_input":"2024-09-28T12:21:04.210656Z","iopub.status.idle":"2024-09-28T12:21:04.215519Z","shell.execute_reply.started":"2024-09-28T12:21:04.210601Z","shell.execute_reply":"2024-09-28T12:21:04.214520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:    \n    \n    test_agg","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.219431Z","iopub.execute_input":"2024-09-28T12:21:04.219764Z","iopub.status.idle":"2024-09-28T12:21:04.224328Z","shell.execute_reply.started":"2024-09-28T12:21:04.219729Z","shell.execute_reply":"2024-09-28T12:21:04.223364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS: \n    \n    train = train.join(train_agg.with_columns(pl.col(\"id\").cast(pl.Categorical)), on='id', how='left')\n    test = test.join(test_agg.with_columns(pl.col(\"id\").cast(pl.Categorical)), on='id', how='left')","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.225453Z","iopub.execute_input":"2024-09-28T12:21:04.225766Z","iopub.status.idle":"2024-09-28T12:21:04.254124Z","shell.execute_reply.started":"2024-09-28T12:21:04.225734Z","shell.execute_reply":"2024-09-28T12:21:04.253302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:    \n    \n    test","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.255318Z","iopub.execute_input":"2024-09-28T12:21:04.255998Z","iopub.status.idle":"2024-09-28T12:21:04.260068Z","shell.execute_reply.started":"2024-09-28T12:21:04.255954Z","shell.execute_reply":"2024-09-28T12:21:04.259102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    # ignore rows with null values in TARGET_COLS\n    train_without_null = train.drop_nulls(subset=TARGET_COLS)\n    X = train_without_null.select(FEATURE_COLS)\n    X_test = test.select(FEATURE_COLS)\n    y = train_without_null.select(TARGET_COLS)\n    y_sii = y.get_column(\"sii\").to_numpy()  # ground truth\n    cat_features = X.select(cs.categorical()).columns","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.261083Z","iopub.execute_input":"2024-09-28T12:21:04.261461Z","iopub.status.idle":"2024-09-28T12:21:04.275227Z","shell.execute_reply.started":"2024-09-28T12:21:04.261418Z","shell.execute_reply":"2024-09-28T12:21:04.274508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n\n    X","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.276341Z","iopub.execute_input":"2024-09-28T12:21:04.276628Z","iopub.status.idle":"2024-09-28T12:21:04.280646Z","shell.execute_reply.started":"2024-09-28T12:21:04.276597Z","shell.execute_reply":"2024-09-28T12:21:04.279689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n\n    X_test","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.281788Z","iopub.execute_input":"2024-09-28T12:21:04.282081Z","iopub.status.idle":"2024-09-28T12:21:04.288062Z","shell.execute_reply.started":"2024-09-28T12:21:04.282050Z","shell.execute_reply":"2024-09-28T12:21:04.287315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    class MultiTargetQWK(MultiTargetCustomMetric):\n        def get_final_error(self, error, weight):\n            return np.sum(error)  # / np.sum(weight)\n\n        def is_max_optimal(self):\n            # if True, the bigger the better\n            return True\n\n        def evaluate(self, approxes, targets, weight):\n            # approxes: 予測値 (shape: [ターゲット数, サンプル数])\n            # targets: 実際の値 (shape: [ターゲット数, サンプル数])\n            # weight: サンプルごとの重み (Noneも可)\n\n            approx = np.clip(approxes[-1], 0, 3).round().astype(int)\n            target = targets[-1]\n\n            qwk = cohen_kappa_score(target, approx, weights=\"quadratic\")\n\n            return qwk, 1\n\n        def get_custom_metric_name(self):\n            return \"MultiTargetQWK\"\n\n\n    class OptimizedRounder:\n        \"\"\"\n        A class for optimizing the rounding of continuous predictions into discrete class labels using Optuna.\n        The optimization process maximizes the Quadratic Weighted Kappa score by learning thresholds that separate\n        continuous predictions into class intervals.\n\n        Args:\n            n_classes (int): The number of discrete class labels.\n            n_trials (int, optional): The number of trials for the Optuna optimization. Defaults to 100.\n\n        Attributes:\n            n_classes (int): The number of discrete class labels.\n            labels (NDArray[np.int_]): An array of class labels from 0 to `n_classes - 1`.\n            n_trials (int): The number of optimization trials.\n            metric (Callable): The Quadratic Weighted Kappa score metric used for optimization.\n            thresholds (List[float]): The optimized thresholds learned after calling `fit()`.\n\n        Methods:\n            fit(y_pred: NDArray[np.float_], y_true: NDArray[np.int_]) -> None:\n                Fits the rounding thresholds based on continuous predictions and ground truth labels.\n\n                Args:\n                    y_pred (NDArray[np.float_]): Continuous predictions that need to be rounded.\n                    y_true (NDArray[np.int_]): Ground truth class labels.\n\n                Returns:\n                    None\n\n            predict(y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n                Predicts discrete class labels by rounding continuous predictions using the fitted thresholds.\n                `fit()` must be called before `predict()`.\n\n                Args:\n                    y_pred (NDArray[np.float_]): Continuous predictions to be rounded.\n\n                Returns:\n                    NDArray[np.int_]: Predicted class labels.\n\n            _normalize(y: NDArray[np.float_]) -> NDArray[np.float_]:\n                Normalizes the continuous values to the range [0, `n_classes - 1`].\n\n                Args:\n                    y (NDArray[np.float_]): Continuous values to be normalized.\n\n                Returns:\n                    NDArray[np.float_]: Normalized values.\n\n        References:\n            - This implementation uses Optuna for threshold optimization.\n            - Quadratic Weighted Kappa is used as the evaluation metric.\n        \"\"\"\n\n        def __init__(self, n_classes: int, n_trials: int = 200):\n            self.n_classes = n_classes\n            self.labels = np.arange(n_classes)\n            self.n_trials = n_trials\n            self.metric = partial(cohen_kappa_score, weights=\"quadratic\")\n\n        def fit(self, y_pred: NDArray[np.float_], y_true: NDArray[np.int_]) -> None:\n            y_pred = self._normalize(y_pred)\n\n            def objective(trial: optuna.Trial) -> float:\n                thresholds = []\n                for i in range(self.n_classes - 1):\n                    low = max(thresholds) if i > 0 else min(self.labels)\n                    high = max(self.labels)\n                    th = trial.suggest_float(f\"threshold_{i}\", low, high)\n                    thresholds.append(th)\n                try:\n                    y_pred_rounded = np.digitize(y_pred, thresholds)\n                except ValueError:\n                    return -100\n                return self.metric(y_true, y_pred_rounded)\n\n            optuna.logging.disable_default_handler()\n            study = optuna.create_study(direction=\"maximize\")\n            study.optimize(\n                objective,\n                n_trials=self.n_trials,\n            )\n            self.thresholds = [study.best_params[f\"threshold_{i}\"] for i in range(self.n_classes - 1)]\n\n        def predict(self, y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n            assert hasattr(self, \"thresholds\"), \"fit() must be called before predict()\"\n            y_pred = self._normalize(y_pred)\n            return np.digitize(y_pred, self.thresholds)\n\n        def _normalize(self, y: NDArray[np.float_]) -> NDArray[np.float_]:\n            # normalize y_pred to [0, n_classes - 1]\n            return (y - y.min()) / (y.max() - y.min()) * (self.n_classes - 1)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.289442Z","iopub.execute_input":"2024-09-28T12:21:04.289825Z","iopub.status.idle":"2024-09-28T12:21:04.309795Z","shell.execute_reply.started":"2024-09-28T12:21:04.289760Z","shell.execute_reply":"2024-09-28T12:21:04.308845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    # setting catboost parameters\n    params = dict(\n        loss_function=\"MultiRMSE\",\n        eval_metric=MultiTargetQWK(),\n        iterations=1 if IS_TEST else 150000,\n        learning_rate=0.05,\n        depth=8,\n        early_stopping_rounds=75,\n    )\n\n    # Cross-validation\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=52)\n    models: list[CatBoostRegressor] = []\n    y_pred = np.full((X.height, len(TARGET_COLS)), fill_value=np.nan)\n    for train_idx, val_idx in skf.split(X, y_sii):\n        X_train: pl.DataFrame\n        X_val: pl.DataFrame\n        y_train: pl.DataFrame\n        y_val: pl.DataFrame\n        X_train, X_val = X[train_idx], X[val_idx]\n        y_train, y_val = y[train_idx], y[val_idx]\n\n        # train model\n        model = CatBoostRegressor(**params)\n        model.fit(\n            X_train.to_pandas(),\n            y_train.to_pandas(),\n            eval_set=(X_val.to_pandas(), y_val.to_pandas()),\n            cat_features=cat_features,\n            verbose=False,\n        )\n        models.append(model)\n\n        # predict\n        y_pred[val_idx] = model.predict(X_val.to_pandas())\n\n    assert np.isnan(y_pred).sum() == 0\n    # Optimize thresholds\n    optimizer = OptimizedRounder(n_classes=4, n_trials=400)\n    y_pred_total = y_pred[:, TARGET_COLS.index(\"PCIAT-PCIAT_Total\")]\n    optimizer.fit(y_pred_total, y_sii)\n    y_pred_rounded = optimizer.predict(y_pred_total)\n\n    # Calculate QWK\n    qwk = cohen_kappa_score(y_sii, y_pred_rounded, weights=\"quadratic\")\n    print(f\"Cross-Validated QWK Score: {qwk}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:04.310966Z","iopub.execute_input":"2024-09-28T12:21:04.311423Z","iopub.status.idle":"2024-09-28T12:21:19.950616Z","shell.execute_reply.started":"2024-09-28T12:21:04.311371Z","shell.execute_reply":"2024-09-28T12:21:19.949620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    feature_importance = np.mean([model.get_feature_importance() for model in models], axis=0)\n    sorted_idx = np.argsort(feature_importance)\n    sorted_idx = sorted_idx[-30:]\n    fig = plt.figure(figsize=(12, 10))\n    plt.barh(range(len(sorted_idx)), feature_importance[sorted_idx], align=\"center\")\n    plt.yticks(range(len(sorted_idx)), np.array(X_test.columns)[sorted_idx])\n    plt.title(\"Feature Importance\")","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:19.951975Z","iopub.execute_input":"2024-09-28T12:21:19.952362Z","iopub.status.idle":"2024-09-28T12:21:20.602940Z","shell.execute_reply.started":"2024-09-28T12:21:19.952317Z","shell.execute_reply":"2024-09-28T12:21:20.601819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    class AvgModel:\n        def __init__(self, models: list[BaseEstimator]):\n            self.models = models\n\n        def predict(self, X: ArrayLike) -> NDArray[np.int_]:\n            preds: list[NDArray[np.int_]] = []\n            for model in self.models:\n                pred = model.predict(X)\n                preds.append(pred)\n\n            return np.mean(preds, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:20.604263Z","iopub.execute_input":"2024-09-28T12:21:20.605036Z","iopub.status.idle":"2024-09-28T12:21:20.612079Z","shell.execute_reply.started":"2024-09-28T12:21:20.604985Z","shell.execute_reply":"2024-09-28T12:21:20.610913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    avg_model = AvgModel(models)\n    test_pred = avg_model.predict(X_test.to_pandas())[:, TARGET_COLS.index(\"PCIAT-PCIAT_Total\")]\n    test_pred_rounded = optimizer.predict(test_pred)\n    test.select(\"id\").with_columns(\n        pl.Series(\"sii\", pl.Series(\"sii\", test_pred_rounded)),\n    ).write_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:21:20.613680Z","iopub.execute_input":"2024-09-28T12:21:20.614039Z","iopub.status.idle":"2024-09-28T12:21:20.646473Z","shell.execute_reply.started":"2024-09-28T12:21:20.614006Z","shell.execute_reply":"2024-09-28T12:21:20.645574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}