{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":197506774,"sourceType":"kernelVersion"}],"dockerImageVersionId":30762,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"LAUNCH_VARIANT,ENSEMBLE_SOLUTIONS = 'option 61',['SOLUTION_11','SOLUTION_14','SOLUTION_15'] ","metadata":{"execution":{"iopub.status.busy":"2024-09-28T10:32:10.757853Z","iopub.execute_input":"2024-09-28T10:32:10.758564Z","iopub.status.idle":"2024-09-28T10:32:10.768134Z","shell.execute_reply.started":"2024-09-28T10:32:10.758524Z","shell.execute_reply":"2024-09-28T10:32:10.767155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    !nvidia-smi","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.774133Z","iopub.execute_input":"2024-09-28T10:32:10.774579Z","iopub.status.idle":"2024-09-28T10:32:10.783277Z","shell.execute_reply.started":"2024-09-28T10:32:10.774546Z","shell.execute_reply":"2024-09-28T10:32:10.782473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    !pip install /kaggle/input/polars-gpu-1-7-1/cupy_cuda12x-13.3.0-cp310-cp310-manylinux2014_x86_64.whl\n    !pip install /kaggle/input/polars-gpu-1-7-1/rmm_cu12-24.8.2-cp310-cp310-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl\n    !pip install /kaggle/input/polars-gpu-1-7-1/cudf_cu12-24.8.3-cp310-cp310-manylinux_2_28_x86_64.whl\n    !pip install /kaggle/input/polars-gpu-1-7-1/polars-1.7.1-cp38-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n    !pip install /kaggle/input/polars-gpu-1-7-1/cudf_polars_cu12-24.8.3-py3-none-any.whl","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.789885Z","iopub.execute_input":"2024-09-28T10:32:10.790221Z","iopub.status.idle":"2024-09-28T10:32:10.803661Z","shell.execute_reply.started":"2024-09-28T10:32:10.790172Z","shell.execute_reply":"2024-09-28T10:32:10.802685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n\n    import warnings\n    from functools import partial\n    from pathlib import Path\n\n    from tqdm import tqdm\n    import matplotlib.pyplot as plt\n    import numpy as np\n    import optuna\n    import polars as pl\n    import polars.selectors as cs\n    from catboost import CatBoostRegressor, MultiTargetCustomMetric\n    from numpy.typing import ArrayLike, NDArray\n    from polars.testing import assert_frame_equal\n    from sklearn.base import BaseEstimator\n    from sklearn.metrics import cohen_kappa_score\n    from sklearn.model_selection import StratifiedKFold\n\n    warnings.filterwarnings(\"ignore\", message=\"Failed to optimize method\")\n\n    DATA_DIR = Path(\"/kaggle/input/child-mind-institute-problematic-internet-use\")\n    TARGET_COLS = [\n        \"PCIAT-PCIAT_01\",\n        \"PCIAT-PCIAT_02\",\n        \"PCIAT-PCIAT_03\",\n        \"PCIAT-PCIAT_04\",\n        \"PCIAT-PCIAT_05\",\n        \"PCIAT-PCIAT_06\",\n        \"PCIAT-PCIAT_07\",\n        \"PCIAT-PCIAT_08\",\n        \"PCIAT-PCIAT_09\",\n        \"PCIAT-PCIAT_10\",\n        \"PCIAT-PCIAT_11\",\n        \"PCIAT-PCIAT_12\",\n        \"PCIAT-PCIAT_13\",\n        \"PCIAT-PCIAT_14\",\n        \"PCIAT-PCIAT_15\",\n        \"PCIAT-PCIAT_16\",\n        \"PCIAT-PCIAT_17\",\n        \"PCIAT-PCIAT_18\",\n        \"PCIAT-PCIAT_19\",\n        \"PCIAT-PCIAT_20\",\n        \"PCIAT-PCIAT_Total\",\n        \"sii\",\n    ]\n\n    FEATURE_COLS = [\n        \"Basic_Demos-Enroll_Season\",\n        \"Basic_Demos-Age\",\n        \"Basic_Demos-Sex\",\n        \"CGAS-Season\",\n        \"CGAS-CGAS_Score\",\n        \"Physical-Season\",\n        \"Physical-BMI\",\n        \"Physical-Height\",\n        \"Physical-Weight\",\n        \"Physical-Waist_Circumference\",\n        \"Physical-Diastolic_BP\",\n        \"Physical-HeartRate\",\n        \"Physical-Systolic_BP\",\n        \"Fitness_Endurance-Season\",\n        \"Fitness_Endurance-Max_Stage\",\n        \"Fitness_Endurance-Time_Mins\",\n        \"Fitness_Endurance-Time_Sec\",\n        \"FGC-Season\",\n        \"FGC-FGC_CU\",\n        \"FGC-FGC_CU_Zone\",\n        \"FGC-FGC_GSND\",\n        \"FGC-FGC_GSND_Zone\",\n        \"FGC-FGC_GSD\",\n        \"FGC-FGC_GSD_Zone\",\n        \"FGC-FGC_PU\",\n        \"FGC-FGC_PU_Zone\",\n        \"FGC-FGC_SRL\",\n        \"FGC-FGC_SRL_Zone\",\n        \"FGC-FGC_SRR\",\n        \"FGC-FGC_SRR_Zone\",\n        \"FGC-FGC_TL\",\n        \"FGC-FGC_TL_Zone\",\n        \"BIA-Season\",\n        \"BIA-BIA_Activity_Level_num\",\n        \"BIA-BIA_BMC\",\n        \"BIA-BIA_BMI\",\n        \"BIA-BIA_BMR\",\n        \"BIA-BIA_DEE\",\n        \"BIA-BIA_ECW\",\n        \"BIA-BIA_FFM\",\n        \"BIA-BIA_FFMI\",\n        \"BIA-BIA_FMI\",\n        \"BIA-BIA_Fat\",\n        \"BIA-BIA_Frame_num\",\n        \"BIA-BIA_ICW\",\n        \"BIA-BIA_LDM\",\n        \"BIA-BIA_LST\",\n        \"BIA-BIA_SMM\",\n        \"BIA-BIA_TBW\",\n        \"PAQ_A-Season\",\n        \"PAQ_A-PAQ_A_Total\",\n        \"PAQ_C-Season\",\n        \"PAQ_C-PAQ_C_Total\",\n        \"SDS-Season\",\n        \"SDS-SDS_Total_Raw\",\n        \"SDS-SDS_Total_T\",\n        \"PreInt_EduHx-Season\",\n        \"PreInt_EduHx-computerinternet_hoursday\",\n\n        # stats features from parquets\n        \"X_min\",\n        \"Y_min\",\n        \"Z_min\",\n        \"enmo_min\",\n        \"anglez_min\",\n        \"light_min\",\n        \"battery_voltage_min\",\n        \"X_mean\",\n        \"Y_mean\",\n        \"Z_mean\",\n        \"enmo_mean\",\n        \"anglez_mean\",\n        \"light_mean\",\n        \"battery_voltage_mean\",\n        \"X_max\",\n        \"Y_max\",\n        \"Z_max\",\n        \"enmo_max\",\n        \"anglez_max\",\n        \"light_max\",\n        \"battery_voltage_max\",\n        \"X_std\",\n        \"Y_std\",\n        \"Z_std\",\n        \"enmo_std\",\n        \"anglez_std\",\n        \"light_std\",\n        \"battery_voltage_std\",\n    ]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.819029Z","iopub.execute_input":"2024-09-28T10:32:10.819307Z","iopub.status.idle":"2024-09-28T10:32:10.831953Z","shell.execute_reply.started":"2024-09-28T10:32:10.819276Z","shell.execute_reply":"2024-09-28T10:32:10.831035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:    \n    \n    # Load data\n    train = pl.read_csv(DATA_DIR / \"train.csv\")\n    test = pl.read_csv(DATA_DIR / \"test.csv\")\n    train_test = pl.concat([train, test], how=\"diagonal\")\n\n    IS_TEST = test.height <= 100\n\n    assert_frame_equal(train, train_test[: train.height].select(train.columns))\n    assert_frame_equal(test, train_test[train.height :].select(test.columns))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.833692Z","iopub.execute_input":"2024-09-28T10:32:10.834053Z","iopub.status.idle":"2024-09-28T10:32:10.844951Z","shell.execute_reply.started":"2024-09-28T10:32:10.834011Z","shell.execute_reply":"2024-09-28T10:32:10.844105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    # Cast string columns to categorical\n    train_test = train_test.with_columns(cs.string().cast(pl.Categorical).fill_null(\"NAN\"))\n    train = train_test[: train.height]\n    test = train_test[train.height :]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.846485Z","iopub.execute_input":"2024-09-28T10:32:10.846822Z","iopub.status.idle":"2024-09-28T10:32:10.854680Z","shell.execute_reply.started":"2024-09-28T10:32:10.846781Z","shell.execute_reply":"2024-09-28T10:32:10.853919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    def split_array(ar, n_group):\n        for i_chunk in range(n_group):\n            yield ar[i_chunk * len(ar) // n_group : (i_chunk + 1) * len(ar) // n_group]\n\n    def agg_parquets(files):\n        cols = [\"X\", \"Y\", \"Z\", \"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n        aggs = []\n        files_chunks = list(split_array(files, 10))\n        for files_tmp in tqdm(files_chunks):\n            if len(files_tmp) == 0:\n                continue\n            dfs = []\n            for file in files_tmp:\n                df = pl.scan_parquet(file)\n                df = df.with_columns(pl.lit(file.parts[-1].split(\"=\")[1]).alias(\"id\"))\n                dfs.append(df)\n            df = pl.concat(dfs)\n            agg = (\n                df.group_by(\"id\")\n                .agg(\n                    [pl.col(c).cast(pl.Float32).min().alias(f\"{c}_min\") for c in cols]\n                    + [pl.col(c).cast(pl.Float32).mean().alias(f\"{c}_mean\") for c in cols]\n                    + [pl.col(c).cast(pl.Float32).max().alias(f\"{c}_max\") for c in cols]\n                    + [pl.col(c).cast(pl.Float32).std().alias(f\"{c}_std\") for c in cols]\n                )\n                .collect(engine=\"gpu\")\n            )\n            aggs.append(agg)\n        return pl.concat(aggs)\n\n    train_agg = agg_parquets(sorted(Path(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\").glob(\"*\")))\n    test_agg = agg_parquets(sorted(Path(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\").glob(\"*\")))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.858519Z","iopub.execute_input":"2024-09-28T10:32:10.858810Z","iopub.status.idle":"2024-09-28T10:32:10.870363Z","shell.execute_reply.started":"2024-09-28T10:32:10.858780Z","shell.execute_reply":"2024-09-28T10:32:10.869639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:    \n    \n    train_agg","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.886500Z","iopub.execute_input":"2024-09-28T10:32:10.886769Z","iopub.status.idle":"2024-09-28T10:32:10.891796Z","shell.execute_reply.started":"2024-09-28T10:32:10.886741Z","shell.execute_reply":"2024-09-28T10:32:10.890993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:    \n    \n    test_agg","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.893421Z","iopub.execute_input":"2024-09-28T10:32:10.893734Z","iopub.status.idle":"2024-09-28T10:32:10.899609Z","shell.execute_reply.started":"2024-09-28T10:32:10.893702Z","shell.execute_reply":"2024-09-28T10:32:10.898770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS: \n    \n    train = train.join(train_agg.with_columns(pl.col(\"id\").cast(pl.Categorical)), on='id', how='left')\n    test = test.join(test_agg.with_columns(pl.col(\"id\").cast(pl.Categorical)), on='id', how='left')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.901168Z","iopub.execute_input":"2024-09-28T10:32:10.901484Z","iopub.status.idle":"2024-09-28T10:32:10.909717Z","shell.execute_reply.started":"2024-09-28T10:32:10.901451Z","shell.execute_reply":"2024-09-28T10:32:10.908788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:    \n    \n    test","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.918908Z","iopub.execute_input":"2024-09-28T10:32:10.919296Z","iopub.status.idle":"2024-09-28T10:32:10.923836Z","shell.execute_reply.started":"2024-09-28T10:32:10.919263Z","shell.execute_reply":"2024-09-28T10:32:10.922880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    # ignore rows with null values in TARGET_COLS\n    train_without_null = train.drop_nulls(subset=TARGET_COLS)\n    X = train_without_null.select(FEATURE_COLS)\n    X_test = test.select(FEATURE_COLS)\n    y = train_without_null.select(TARGET_COLS)\n    y_sii = y.get_column(\"sii\").to_numpy()  # ground truth\n    cat_features = X.select(cs.categorical()).columns","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.930932Z","iopub.execute_input":"2024-09-28T10:32:10.931623Z","iopub.status.idle":"2024-09-28T10:32:10.937697Z","shell.execute_reply.started":"2024-09-28T10:32:10.931578Z","shell.execute_reply":"2024-09-28T10:32:10.936810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n\n    X","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:10.948127Z","iopub.execute_input":"2024-09-28T10:32:10.948408Z","iopub.status.idle":"2024-09-28T10:32:10.953464Z","shell.execute_reply.started":"2024-09-28T10:32:10.948368Z","shell.execute_reply":"2024-09-28T10:32:10.952623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n\n    X_test","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:10.960792Z","iopub.execute_input":"2024-09-28T10:32:10.961102Z","iopub.status.idle":"2024-09-28T10:32:10.965443Z","shell.execute_reply.started":"2024-09-28T10:32:10.961069Z","shell.execute_reply":"2024-09-28T10:32:10.964594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    class MultiTargetQWK(MultiTargetCustomMetric):\n        def get_final_error(self, error, weight):\n            return np.sum(error)  # / np.sum(weight)\n\n        def is_max_optimal(self):\n            # if True, the bigger the better\n            return True\n\n        def evaluate(self, approxes, targets, weight):\n            # approxes: 予測値 (shape: [ターゲット数, サンプル数])\n            # targets: 実際の値 (shape: [ターゲット数, サンプル数])\n            # weight: サンプルごとの重み (Noneも可)\n\n            approx = np.clip(approxes[-1], 0, 3).round().astype(int)\n            target = targets[-1]\n\n            qwk = cohen_kappa_score(target, approx, weights=\"quadratic\")\n\n            return qwk, 1\n\n        def get_custom_metric_name(self):\n            return \"MultiTargetQWK\"\n\n\n    class OptimizedRounder:\n        \"\"\"\n        A class for optimizing the rounding of continuous predictions into discrete class labels using Optuna.\n        The optimization process maximizes the Quadratic Weighted Kappa score by learning thresholds that separate\n        continuous predictions into class intervals.\n\n        Args:\n            n_classes (int): The number of discrete class labels.\n            n_trials (int, optional): The number of trials for the Optuna optimization. Defaults to 100.\n\n        Attributes:\n            n_classes (int): The number of discrete class labels.\n            labels (NDArray[np.int_]): An array of class labels from 0 to `n_classes - 1`.\n            n_trials (int): The number of optimization trials.\n            metric (Callable): The Quadratic Weighted Kappa score metric used for optimization.\n            thresholds (List[float]): The optimized thresholds learned after calling `fit()`.\n\n        Methods:\n            fit(y_pred: NDArray[np.float_], y_true: NDArray[np.int_]) -> None:\n                Fits the rounding thresholds based on continuous predictions and ground truth labels.\n\n                Args:\n                    y_pred (NDArray[np.float_]): Continuous predictions that need to be rounded.\n                    y_true (NDArray[np.int_]): Ground truth class labels.\n\n                Returns:\n                    None\n\n            predict(y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n                Predicts discrete class labels by rounding continuous predictions using the fitted thresholds.\n                `fit()` must be called before `predict()`.\n\n                Args:\n                    y_pred (NDArray[np.float_]): Continuous predictions to be rounded.\n\n                Returns:\n                    NDArray[np.int_]: Predicted class labels.\n\n            _normalize(y: NDArray[np.float_]) -> NDArray[np.float_]:\n                Normalizes the continuous values to the range [0, `n_classes - 1`].\n\n                Args:\n                    y (NDArray[np.float_]): Continuous values to be normalized.\n\n                Returns:\n                    NDArray[np.float_]: Normalized values.\n\n        References:\n            - This implementation uses Optuna for threshold optimization.\n            - Quadratic Weighted Kappa is used as the evaluation metric.\n        \"\"\"\n\n        def __init__(self, n_classes: int, n_trials: int = 100):\n            self.n_classes = n_classes\n            self.labels = np.arange(n_classes)\n            self.n_trials = n_trials\n            self.metric = partial(cohen_kappa_score, weights=\"quadratic\")\n\n        def fit(self, y_pred: NDArray[np.float_], y_true: NDArray[np.int_]) -> None:\n            y_pred = self._normalize(y_pred)\n\n            def objective(trial: optuna.Trial) -> float:\n                thresholds = []\n                for i in range(self.n_classes - 1):\n                    low = max(thresholds) if i > 0 else min(self.labels)\n                    high = max(self.labels)\n                    th = trial.suggest_float(f\"threshold_{i}\", low, high)\n                    thresholds.append(th)\n                try:\n                    y_pred_rounded = np.digitize(y_pred, thresholds)\n                except ValueError:\n                    return -100\n                return self.metric(y_true, y_pred_rounded)\n\n            optuna.logging.disable_default_handler()\n            study = optuna.create_study(direction=\"maximize\")\n            study.optimize(\n                objective,\n                n_trials=self.n_trials,\n            )\n            self.thresholds = [study.best_params[f\"threshold_{i}\"] for i in range(self.n_classes - 1)]\n\n        def predict(self, y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n            assert hasattr(self, \"thresholds\"), \"fit() must be called before predict()\"\n            y_pred = self._normalize(y_pred)\n            return np.digitize(y_pred, self.thresholds)\n\n        def _normalize(self, y: NDArray[np.float_]) -> NDArray[np.float_]:\n            # normalize y_pred to [0, n_classes - 1]\n            return (y - y.min()) / (y.max() - y.min()) * (self.n_classes - 1)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:10.977853Z","iopub.execute_input":"2024-09-28T10:32:10.978127Z","iopub.status.idle":"2024-09-28T10:32:10.996653Z","shell.execute_reply.started":"2024-09-28T10:32:10.978097Z","shell.execute_reply":"2024-09-28T10:32:10.995768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    # setting catboost parameters\n    params = dict(\n        loss_function=\"MultiRMSE\",\n        eval_metric=MultiTargetQWK(),\n        iterations=1 if IS_TEST else 100000,\n        learning_rate=0.1,\n        depth=5,\n        early_stopping_rounds=50,\n    )\n\n    # Cross-validation\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=52)\n    models: list[CatBoostRegressor] = []\n    y_pred = np.full((X.height, len(TARGET_COLS)), fill_value=np.nan)\n    for train_idx, val_idx in skf.split(X, y_sii):\n        X_train: pl.DataFrame\n        X_val: pl.DataFrame\n        y_train: pl.DataFrame\n        y_val: pl.DataFrame\n        X_train, X_val = X[train_idx], X[val_idx]\n        y_train, y_val = y[train_idx], y[val_idx]\n\n        # train model\n        model = CatBoostRegressor(**params)\n        model.fit(\n            X_train.to_pandas(),\n            y_train.to_pandas(),\n            eval_set=(X_val.to_pandas(), y_val.to_pandas()),\n            cat_features=cat_features,\n            verbose=False,\n        )\n        models.append(model)\n\n        # predict\n        y_pred[val_idx] = model.predict(X_val.to_pandas())\n\n    assert np.isnan(y_pred).sum() == 0\n    # Optimize thresholds\n    optimizer = OptimizedRounder(n_classes=4, n_trials=300)\n    y_pred_total = y_pred[:, TARGET_COLS.index(\"PCIAT-PCIAT_Total\")]\n    optimizer.fit(y_pred_total, y_sii)\n    y_pred_rounded = optimizer.predict(y_pred_total)\n\n    # Calculate QWK\n    qwk = cohen_kappa_score(y_sii, y_pred_rounded, weights=\"quadratic\")\n    print(f\"Cross-Validated QWK Score: {qwk}\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:10.998327Z","iopub.execute_input":"2024-09-28T10:32:10.998639Z","iopub.status.idle":"2024-09-28T10:32:11.011791Z","shell.execute_reply.started":"2024-09-28T10:32:10.998600Z","shell.execute_reply":"2024-09-28T10:32:11.010936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    feature_importance = np.mean([model.get_feature_importance() for model in models], axis=0)\n    sorted_idx = np.argsort(feature_importance)\n    sorted_idx = sorted_idx[-30:]\n    fig = plt.figure(figsize=(12, 10))\n    plt.barh(range(len(sorted_idx)), feature_importance[sorted_idx], align=\"center\")\n    plt.yticks(range(len(sorted_idx)), np.array(X_test.columns)[sorted_idx])\n    plt.title(\"Feature Importance\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.013221Z","iopub.execute_input":"2024-09-28T10:32:11.014052Z","iopub.status.idle":"2024-09-28T10:32:11.024950Z","shell.execute_reply.started":"2024-09-28T10:32:11.014008Z","shell.execute_reply":"2024-09-28T10:32:11.024011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    class AvgModel:\n        def __init__(self, models: list[BaseEstimator]):\n            self.models = models\n\n        def predict(self, X: ArrayLike) -> NDArray[np.int_]:\n            preds: list[NDArray[np.int_]] = []\n            for model in self.models:\n                pred = model.predict(X)\n                preds.append(pred)\n\n            return np.mean(preds, axis=0)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.026995Z","iopub.execute_input":"2024-09-28T10:32:11.027298Z","iopub.status.idle":"2024-09-28T10:32:11.036262Z","shell.execute_reply.started":"2024-09-28T10:32:11.027266Z","shell.execute_reply":"2024-09-28T10:32:11.035415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_1' in ENSEMBLE_SOLUTIONS:\n    \n    avg_model = AvgModel(models)\n    test_pred = avg_model.predict(X_test.to_pandas())[:, TARGET_COLS.index(\"PCIAT-PCIAT_Total\")]\n    test_pred_rounded = optimizer.predict(test_pred)\n    test.select(\"id\").with_columns(\n        pl.Series(\"sii\", pl.Series(\"sii\", test_pred_rounded)),\n    ).write_csv(\"submission_1.csv\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.037284Z","iopub.execute_input":"2024-09-28T10:32:11.037614Z","iopub.status.idle":"2024-09-28T10:32:11.045896Z","shell.execute_reply.started":"2024-09-28T10:32:11.037573Z","shell.execute_reply":"2024-09-28T10:32:11.044953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_2' in ENSEMBLE_SOLUTIONS: pass","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.047708Z","iopub.execute_input":"2024-09-28T10:32:11.048429Z","iopub.status.idle":"2024-09-28T10:32:11.055155Z","shell.execute_reply.started":"2024-09-28T10:32:11.048386Z","shell.execute_reply":"2024-09-28T10:32:11.054195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import warnings\n# from functools import partial\n# from pathlib import Path\n\n# import matplotlib.pyplot as plt\n# import numpy as np\n# import optuna\n# import polars as pl\n# import polars.selectors as cs\n# from catboost import CatBoostRegressor, MultiTargetCustomMetric\n# from numpy.typing import ArrayLike, NDArray\n# from polars.testing import assert_frame_equal\n# from sklearn.base import BaseEstimator\n# from sklearn.metrics import cohen_kappa_score\n# from sklearn.model_selection import StratifiedKFold\n\n# warnings.filterwarnings(\"ignore\", message=\"Failed to optimize method\")\n\n# DATA_DIR = Path(\"/kaggle/input/child-mind-institute-problematic-internet-use\")\n# TARGET_COLS = [\n#     \"PCIAT-PCIAT_01\",\n#     \"PCIAT-PCIAT_02\",\n#     \"PCIAT-PCIAT_03\",\n#     \"PCIAT-PCIAT_04\",\n#     \"PCIAT-PCIAT_05\",\n#     \"PCIAT-PCIAT_06\",\n#     \"PCIAT-PCIAT_07\",\n#     \"PCIAT-PCIAT_08\",\n#     \"PCIAT-PCIAT_09\",\n#     \"PCIAT-PCIAT_10\",\n#     \"PCIAT-PCIAT_11\",\n#     \"PCIAT-PCIAT_12\",\n#     \"PCIAT-PCIAT_13\",\n#     \"PCIAT-PCIAT_14\",\n#     \"PCIAT-PCIAT_15\",\n#     \"PCIAT-PCIAT_16\",\n#     \"PCIAT-PCIAT_17\",\n#     \"PCIAT-PCIAT_18\",\n#     \"PCIAT-PCIAT_19\",\n#     \"PCIAT-PCIAT_20\",\n#     \"PCIAT-PCIAT_Total\",\n#     \"sii\",\n# ]\n\n# FEATURE_COLS = [\n#     \"Basic_Demos-Enroll_Season\",\n#     \"Basic_Demos-Age\",\n#     \"Basic_Demos-Sex\",\n#     \"CGAS-Season\",\n#     \"CGAS-CGAS_Score\",\n#     \"Physical-Season\",\n#     \"Physical-BMI\",\n#     \"Physical-Height\",\n#     \"Physical-Weight\",\n#     \"Physical-Waist_Circumference\",\n#     \"Physical-Diastolic_BP\",\n#     \"Physical-HeartRate\",\n#     \"Physical-Systolic_BP\",\n#     \"Fitness_Endurance-Season\",\n#     \"Fitness_Endurance-Max_Stage\",\n#     \"Fitness_Endurance-Time_Mins\",\n#     \"Fitness_Endurance-Time_Sec\",\n#     \"FGC-Season\",\n#     \"FGC-FGC_CU\",\n#     \"FGC-FGC_CU_Zone\",\n#     \"FGC-FGC_GSND\",\n#     \"FGC-FGC_GSND_Zone\",\n#     \"FGC-FGC_GSD\",\n#     \"FGC-FGC_GSD_Zone\",\n#     \"FGC-FGC_PU\",\n#     \"FGC-FGC_PU_Zone\",\n#     \"FGC-FGC_SRL\",\n#     \"FGC-FGC_SRL_Zone\",\n#     \"FGC-FGC_SRR\",\n#     \"FGC-FGC_SRR_Zone\",\n#     \"FGC-FGC_TL\",\n#     \"FGC-FGC_TL_Zone\",\n#     \"BIA-Season\",\n#     \"BIA-BIA_Activity_Level_num\",\n#     \"BIA-BIA_BMC\",\n#     \"BIA-BIA_BMI\",\n#     \"BIA-BIA_BMR\",\n#     \"BIA-BIA_DEE\",\n#     \"BIA-BIA_ECW\",\n#     \"BIA-BIA_FFM\",\n#     \"BIA-BIA_FFMI\",\n#     \"BIA-BIA_FMI\",\n#     \"BIA-BIA_Fat\",\n#     \"BIA-BIA_Frame_num\",\n#     \"BIA-BIA_ICW\",\n#     \"BIA-BIA_LDM\",\n#     \"BIA-BIA_LST\",\n#     \"BIA-BIA_SMM\",\n#     \"BIA-BIA_TBW\",\n#     \"PAQ_A-Season\",\n#     \"PAQ_A-PAQ_A_Total\",\n#     \"PAQ_C-Season\",\n#     \"PAQ_C-PAQ_C_Total\",\n#     \"SDS-Season\",\n#     \"SDS-SDS_Total_Raw\",\n#     \"SDS-SDS_Total_T\",\n#     \"PreInt_EduHx-Season\",\n#     \"PreInt_EduHx-computerinternet_hoursday\",\n# ]","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.061148Z","iopub.execute_input":"2024-09-28T10:32:11.061411Z","iopub.status.idle":"2024-09-28T10:32:11.069495Z","shell.execute_reply.started":"2024-09-28T10:32:11.061381Z","shell.execute_reply":"2024-09-28T10:32:11.068580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Load data\n# train = pl.read_csv(DATA_DIR / \"train.csv\")\n# test = pl.read_csv(DATA_DIR / \"test.csv\")\n# train_test = pl.concat([train, test], how=\"diagonal\")\n\n# IS_TEST = test.height <= 100\n\n# assert_frame_equal(train, train_test[: train.height].select(train.columns))\n# assert_frame_equal(test, train_test[train.height :].select(test.columns))","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.073341Z","iopub.execute_input":"2024-09-28T10:32:11.073624Z","iopub.status.idle":"2024-09-28T10:32:11.082779Z","shell.execute_reply.started":"2024-09-28T10:32:11.073593Z","shell.execute_reply":"2024-09-28T10:32:11.081924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Cast string columns to categorical\n# train_test = train_test.with_columns(cs.string().cast(pl.Categorical).fill_null(\"NAN\"))\n# train = train_test[: train.height]\n# test = train_test[train.height :]\n\n# # ignore rows with null values in TARGET_COLS\n# train_without_null = train_test.drop_nulls(subset=TARGET_COLS)\n# X = train_without_null.select(FEATURE_COLS)\n# X_test = test.select(FEATURE_COLS)\n# y = train_without_null.select(TARGET_COLS)\n# y_sii = y.get_column(\"sii\").to_numpy()  # ground truth\n# cat_features = X.select(cs.categorical()).columns","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.084283Z","iopub.execute_input":"2024-09-28T10:32:11.084628Z","iopub.status.idle":"2024-09-28T10:32:11.096186Z","shell.execute_reply.started":"2024-09-28T10:32:11.084593Z","shell.execute_reply":"2024-09-28T10:32:11.095293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class MultiTargetQWK(MultiTargetCustomMetric):\n#     def get_final_error(self, error, weight):\n#         return np.sum(error)  # / np.sum(weight)\n\n#     def is_max_optimal(self):\n#         # if True, the bigger the better\n#         return True\n\n#     def evaluate(self, approxes, targets, weight):\n#         # approxes: 予測値 (shape: [ターゲット数, サンプル数])\n#         # targets: 実際の値 (shape: [ターゲット数, サンプル数])\n#         # weight: サンプルごとの重み (Noneも可)\n\n#         approx = np.clip(approxes[-1], 0, 3).round().astype(int)\n#         target = targets[-1]\n\n#         qwk = cohen_kappa_score(target, approx, weights=\"quadratic\")\n\n#         return qwk, 1\n\n#     def get_custom_metric_name(self):\n#         return \"MultiTargetQWK\"\n\n\n# class OptimizedRounder:\n#     \"\"\"\n#     A class for optimizing the rounding of continuous predictions into discrete class labels using Optuna.\n#     The optimization process maximizes the Quadratic Weighted Kappa score by learning thresholds that separate\n#     continuous predictions into class intervals.\n\n#     Args:\n#         n_classes (int): The number of discrete class labels.\n#         n_trials (int, optional): The number of trials for the Optuna optimization. Defaults to 100.\n\n#     Attributes:\n#         n_classes (int): The number of discrete class labels.\n#         labels (NDArray[np.int_]): An array of class labels from 0 to `n_classes - 1`.\n#         n_trials (int): The number of optimization trials.\n#         metric (Callable): The Quadratic Weighted Kappa score metric used for optimization.\n#         thresholds (List[float]): The optimized thresholds learned after calling `fit()`.\n\n#     Methods:\n#         fit(y_pred: NDArray[np.float_], y_true: NDArray[np.int_]) -> None:\n#             Fits the rounding thresholds based on continuous predictions and ground truth labels.\n\n#             Args:\n#                 y_pred (NDArray[np.float_]): Continuous predictions that need to be rounded.\n#                 y_true (NDArray[np.int_]): Ground truth class labels.\n\n#             Returns:\n#                 None\n\n#         predict(y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n#             Predicts discrete class labels by rounding continuous predictions using the fitted thresholds.\n#             `fit()` must be called before `predict()`.\n\n#             Args:\n#                 y_pred (NDArray[np.float_]): Continuous predictions to be rounded.\n\n#             Returns:\n#                 NDArray[np.int_]: Predicted class labels.\n\n#         _normalize(y: NDArray[np.float_]) -> NDArray[np.float_]:\n#             Normalizes the continuous values to the range [0, `n_classes - 1`].\n\n#             Args:\n#                 y (NDArray[np.float_]): Continuous values to be normalized.\n\n#             Returns:\n#                 NDArray[np.float_]: Normalized values.\n\n#     References:\n#         - This implementation uses Optuna for threshold optimization.\n#         - Quadratic Weighted Kappa is used as the evaluation metric.\n#     \"\"\"\n\n#     def __init__(self, n_classes: int, n_trials: int = 100):\n#         self.n_classes = n_classes\n#         self.labels = np.arange(n_classes)\n#         self.n_trials = n_trials\n#         self.metric = partial(cohen_kappa_score, weights=\"quadratic\")\n\n#     def fit(self, y_pred: NDArray[np.float_], y_true: NDArray[np.int_]) -> None:\n#         y_pred = self._normalize(y_pred)\n\n#         def objective(trial: optuna.Trial) -> float:\n#             thresholds = []\n#             for i in range(self.n_classes - 1):\n#                 low = max(thresholds) if i > 0 else min(self.labels)\n#                 high = max(self.labels)\n#                 th = trial.suggest_float(f\"threshold_{i}\", low, high)\n#                 thresholds.append(th)\n#             try:\n#                 y_pred_rounded = np.digitize(y_pred, thresholds)\n#             except ValueError:\n#                 return -100\n#             return self.metric(y_true, y_pred_rounded)\n\n#         optuna.logging.disable_default_handler()\n#         study = optuna.create_study(direction=\"maximize\")\n#         study.optimize(\n#             objective,\n#             n_trials=self.n_trials,\n#         )\n#         self.thresholds = [study.best_params[f\"threshold_{i}\"] for i in range(self.n_classes - 1)]\n\n#     def predict(self, y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n#         assert hasattr(self, \"thresholds\"), \"fit() must be called before predict()\"\n#         y_pred = self._normalize(y_pred)\n#         return np.digitize(y_pred, self.thresholds)\n\n#     def _normalize(self, y: NDArray[np.float_]) -> NDArray[np.float_]:\n#         # normalize y_pred to [0, n_classes - 1]\n#         return (y - y.min()) / (y.max() - y.min()) * (self.n_classes - 1)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.100500Z","iopub.execute_input":"2024-09-28T10:32:11.100782Z","iopub.status.idle":"2024-09-28T10:32:11.108867Z","shell.execute_reply.started":"2024-09-28T10:32:11.100750Z","shell.execute_reply":"2024-09-28T10:32:11.108004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # setting catboost parameters\n# params = dict(\n#     loss_function=\"MultiRMSE\",\n#     eval_metric=MultiTargetQWK(),\n#     iterations=1 if IS_TEST else 100000,\n#     learning_rate=0.1,\n#     depth=5,\n#     early_stopping_rounds=50,\n# )\n\n# # Cross-validation\n# skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=52)\n# models: list[CatBoostRegressor] = []\n# y_pred = np.full((X.height, len(TARGET_COLS)), fill_value=np.nan)\n# for train_idx, val_idx in skf.split(X, y_sii):\n#     X_train: pl.DataFrame\n#     X_val: pl.DataFrame\n#     y_train: pl.DataFrame\n#     y_val: pl.DataFrame\n#     X_train, X_val = X[train_idx], X[val_idx]\n#     y_train, y_val = y[train_idx], y[val_idx]\n\n#     # train model\n#     model = CatBoostRegressor(**params)\n#     model.fit(\n#         X_train.to_pandas(),\n#         y_train.to_pandas(),\n#         eval_set=(X_val.to_pandas(), y_val.to_pandas()),\n#         cat_features=cat_features,\n#         verbose=False,\n#     )\n#     models.append(model)\n\n#     # predict\n#     y_pred[val_idx] = model.predict(X_val.to_pandas())\n\n# assert np.isnan(y_pred).sum() == 0\n# # Optimize thresholds\n# optimizer = OptimizedRounder(n_classes=4, n_trials=300)\n# y_pred_total = y_pred[:, TARGET_COLS.index(\"PCIAT-PCIAT_Total\")]\n# optimizer.fit(y_pred_total, y_sii)\n# y_pred_rounded = optimizer.predict(y_pred_total)\n\n# # Calculate QWK\n# qwk = cohen_kappa_score(y_sii, y_pred_rounded, weights=\"quadratic\")\n# print(f\"Cross-Validated QWK Score: {qwk}\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.110608Z","iopub.execute_input":"2024-09-28T10:32:11.111001Z","iopub.status.idle":"2024-09-28T10:32:11.123577Z","shell.execute_reply.started":"2024-09-28T10:32:11.110939Z","shell.execute_reply":"2024-09-28T10:32:11.122640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature_importance = np.mean([model.get_feature_importance() for model in models], axis=0)\n# sorted_idx = np.argsort(feature_importance)\n# fig = plt.figure(figsize=(12, 10))\n# plt.barh(range(len(sorted_idx)), feature_importance[sorted_idx], align=\"center\")\n# plt.yticks(range(len(sorted_idx)), np.array(X_test.columns)[sorted_idx])\n# plt.title(\"Feature Importance\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.127298Z","iopub.execute_input":"2024-09-28T10:32:11.127580Z","iopub.status.idle":"2024-09-28T10:32:11.136264Z","shell.execute_reply.started":"2024-09-28T10:32:11.127532Z","shell.execute_reply":"2024-09-28T10:32:11.135526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class AvgModel:\n#     def __init__(self, models: list[BaseEstimator]):\n#         self.models = models\n\n#     def predict(self, X: ArrayLike) -> NDArray[np.int_]:\n#         preds: list[NDArray[np.int_]] = []\n#         for model in self.models:\n#             pred = model.predict(X)\n#             preds.append(pred)\n\n#         return np.mean(preds, axis=0)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.145237Z","iopub.execute_input":"2024-09-28T10:32:11.145758Z","iopub.status.idle":"2024-09-28T10:32:11.149409Z","shell.execute_reply.started":"2024-09-28T10:32:11.145726Z","shell.execute_reply":"2024-09-28T10:32:11.148548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# avg_model = AvgModel(models)\n# test_pred = avg_model.predict(X_test.to_pandas())[:, TARGET_COLS.index(\"PCIAT-PCIAT_Total\")]\n# test_pred_rounded = optimizer.predict(test_pred)\n# test.select(\"id\").with_columns(\n#     pl.Series(\"sii\", pl.Series(\"sii\", test_pred_rounded)),\n# ).write_csv(\"submission_2.csv\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.167551Z","iopub.execute_input":"2024-09-28T10:32:11.167843Z","iopub.status.idle":"2024-09-28T10:32:11.171831Z","shell.execute_reply.started":"2024-09-28T10:32:11.167811Z","shell.execute_reply":"2024-09-28T10:32:11.170989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_3' in ENSEMBLE_SOLUTIONS: pass","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.196268Z","iopub.execute_input":"2024-09-28T10:32:11.196542Z","iopub.status.idle":"2024-09-28T10:32:11.201108Z","shell.execute_reply.started":"2024-09-28T10:32:11.196511Z","shell.execute_reply":"2024-09-28T10:32:11.199929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# import polars as pl\n# import pandas as pd\n# from pathlib import Path\n\n# import gc\n# import os\n# import sys\n\n# import re\n\n# from tqdm import tqdm\n# from IPython.display import clear_output\n\n# import warnings\n# warnings.filterwarnings('ignore')\n# pd.options.display.max_columns = None\n\n# import lightgbm as lgb\n# from catboost import CatBoostRegressor\n# from sklearn.ensemble import VotingRegressor\n# from sklearn.model_selection import *\n# from sklearn.metrics import *\n# from xgboost import XGBRegressor\n\n\n# SEED = 42\n# n_splits = 5","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.213849Z","iopub.execute_input":"2024-09-28T10:32:11.214351Z","iopub.status.idle":"2024-09-28T10:32:11.218307Z","shell.execute_reply.started":"2024-09-28T10:32:11.214319Z","shell.execute_reply":"2024-09-28T10:32:11.217417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n\n# train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n# test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n# featuresCols = ['id', 'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n#        'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n#        'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n#        'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n#        'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n#        'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n#        'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n#        'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n#        'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n#        'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n#        'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n#        'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n#        'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n#        'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n#        'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n#        'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n#        'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n#        'PreInt_EduHx-computerinternet_hoursday','sii']\n\n# train = train[featuresCols]\n# train = train.dropna(subset='sii')\n\n# def update(df):\n#     cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n#      'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n    \n#     id_col = ['id']\n    \n#     for c in id_col:\n#         df[c] = df[c].astype('category')\n\n#     for c in cat_c : \n#         df[c] = df[c].fillna('Missing')\n#         df[c] = df[c].astype('category')\n        \n#     return df\n        \n# train = update(train)\n# test = update(test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.229567Z","iopub.execute_input":"2024-09-28T10:32:11.229821Z","iopub.status.idle":"2024-09-28T10:32:11.235777Z","shell.execute_reply.started":"2024-09-28T10:32:11.229792Z","shell.execute_reply":"2024-09-28T10:32:11.234866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.242985Z","iopub.execute_input":"2024-09-28T10:32:11.243362Z","iopub.status.idle":"2024-09-28T10:32:11.246934Z","shell.execute_reply.started":"2024-09-28T10:32:11.243330Z","shell.execute_reply":"2024-09-28T10:32:11.246010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.257240Z","iopub.execute_input":"2024-09-28T10:32:11.257494Z","iopub.status.idle":"2024-09-28T10:32:11.260989Z","shell.execute_reply.started":"2024-09-28T10:32:11.257465Z","shell.execute_reply":"2024-09-28T10:32:11.260154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def quadratic_weighted_kappa(y_true, y_pred):\n#     return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# X = train.drop(['sii'], axis=1)\n# y = train['sii']\n\n# SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n# def TrainML(model_class, test_data):\n#     train_kappa_scores = []\n#     test_kappa_scores = []\n#     all_test_preds = []\n\n#     for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n\n#         model = model_class \n        \n#         X_train, X_test = X.iloc[train_idx], X.iloc[test_idx]\n#         y_train, y_test = y.iloc[train_idx], y.iloc[test_idx]\n\n#         model.fit(X_train, y_train)\n\n#         y_train_pred = model.predict(X_train)\n#         y_test_pred = model.predict(X_test)\n\n#         all_test_preds.append(model.predict(test_data))\n\n#         train_kappa = quadratic_weighted_kappa(y_train, y_train_pred)\n#         test_kappa = quadratic_weighted_kappa(y_test, y_test_pred)\n\n#         train_kappa_scores.append(train_kappa)\n#         test_kappa_scores.append(test_kappa)\n\n#         print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Test QWK: {test_kappa:.4f}\")\n\n#         clear_output(wait=True)\n\n#     print(\"\\n--- Final Mean Scores ---\")\n#     print(f\"Mean Train QWK: {np.mean(train_kappa_scores):.4f}\")\n#     print(f\"Mean Test QWK: {np.mean(test_kappa_scores):.4f}\")\n\n#     mean_test_preds = np.mean(all_test_preds, axis=0)\n\n#     submission = pd.DataFrame({\n#         'id': test_data['id'], \n#         'sii': np.round(mean_test_preds).astype(int)\n#     })\n\n#     return submission","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.271809Z","iopub.execute_input":"2024-09-28T10:32:11.272318Z","iopub.status.idle":"2024-09-28T10:32:11.277453Z","shell.execute_reply.started":"2024-09-28T10:32:11.272286Z","shell.execute_reply":"2024-09-28T10:32:11.276551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Light = lgb.LGBMClassifier(random_state=SEED, verbose=-1)\n# Submission = TrainML(Light,test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.284879Z","iopub.execute_input":"2024-09-28T10:32:11.285532Z","iopub.status.idle":"2024-09-28T10:32:11.290050Z","shell.execute_reply.started":"2024-09-28T10:32:11.285496Z","shell.execute_reply":"2024-09-28T10:32:11.289302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission.to_csv('submission_3.csv', index=False)\n# Submission.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.300498Z","iopub.execute_input":"2024-09-28T10:32:11.300752Z","iopub.status.idle":"2024-09-28T10:32:11.304392Z","shell.execute_reply.started":"2024-09-28T10:32:11.300723Z","shell.execute_reply":"2024-09-28T10:32:11.303518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_4' in ENSEMBLE_SOLUTIONS: pass","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.315760Z","iopub.execute_input":"2024-09-28T10:32:11.316065Z","iopub.status.idle":"2024-09-28T10:32:11.319937Z","shell.execute_reply.started":"2024-09-28T10:32:11.316034Z","shell.execute_reply":"2024-09-28T10:32:11.319082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_5' in ENSEMBLE_SOLUTIONS: pass","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.332284Z","iopub.execute_input":"2024-09-28T10:32:11.332555Z","iopub.status.idle":"2024-09-28T10:32:11.337664Z","shell.execute_reply.started":"2024-09-28T10:32:11.332525Z","shell.execute_reply":"2024-09-28T10:32:11.336926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_7' in ENSEMBLE_SOLUTIONS: pass","metadata":{"_kg_hide-input":false,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.346275Z","iopub.execute_input":"2024-09-28T10:32:11.346550Z","iopub.status.idle":"2024-09-28T10:32:11.350505Z","shell.execute_reply.started":"2024-09-28T10:32:11.346520Z","shell.execute_reply":"2024-09-28T10:32:11.349481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_8' in ENSEMBLE_SOLUTIONS: \n    \n    import numpy as np\n    import polars as pl\n    import pandas as pd\n    from sklearn.base import clone\n    from copy import deepcopy\n    import optuna\n    from scipy.optimize import minimize\n\n    import re\n    from colorama import Fore, Style\n\n    from tqdm import tqdm\n    from IPython.display import clear_output\n\n    import warnings\n    warnings.filterwarnings('ignore')\n    pd.options.display.max_columns = None\n\n    import lightgbm as lgb\n    from catboost import CatBoostRegressor, CatBoostClassifier\n    from xgboost import XGBRegressor\n    from sklearn.ensemble import VotingRegressor\n    from sklearn.model_selection import *\n    from sklearn.metrics import *\n\n    SEED = 42\n    n_splits = 5","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.360138Z","iopub.execute_input":"2024-09-28T10:32:11.360405Z","iopub.status.idle":"2024-09-28T10:32:11.366814Z","shell.execute_reply.started":"2024-09-28T10:32:11.360375Z","shell.execute_reply":"2024-09-28T10:32:11.365998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_8' in ENSEMBLE_SOLUTIONS:\n\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n    train = train.drop('id',axis=1)\n    test = test.drop('id',axis=1)\n\n    featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\n\n    train = train[featuresCols]\n    train = train.dropna(subset='sii')\n\n    cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\n    def update(df):\n        global cat_c\n        for c in cat_c : \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n\n        return df\n\n    train = update(train)\n    test = update(test)\n\n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n\n    for col in cat_c:\n\n        mapping = create_mapping(col, train)\n        mappingTe = create_mapping(col, test)\n\n        train[col] = train[col].replace(mapping).astype(int)\n        test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.378320Z","iopub.execute_input":"2024-09-28T10:32:11.378862Z","iopub.status.idle":"2024-09-28T10:32:11.390428Z","shell.execute_reply.started":"2024-09-28T10:32:11.378830Z","shell.execute_reply":"2024-09-28T10:32:11.389474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_8' in ENSEMBLE_SOLUTIONS:\n\n    train.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.391867Z","iopub.execute_input":"2024-09-28T10:32:11.392215Z","iopub.status.idle":"2024-09-28T10:32:11.404328Z","shell.execute_reply.started":"2024-09-28T10:32:11.392171Z","shell.execute_reply":"2024-09-28T10:32:11.403524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_8' in ENSEMBLE_SOLUTIONS:\n\n    test.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.405840Z","iopub.execute_input":"2024-09-28T10:32:11.406484Z","iopub.status.idle":"2024-09-28T10:32:11.414282Z","shell.execute_reply.started":"2024-09-28T10:32:11.406450Z","shell.execute_reply":"2024-09-28T10:32:11.413443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_8' in ENSEMBLE_SOLUTIONS:\n\n    def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n    def threshold_Rounder(oof_non_rounded, thresholds):\n        return np.where(oof_non_rounded < thresholds[0], 0,\n                        np.where(oof_non_rounded < thresholds[1], 1,\n                                 np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n    def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n        rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_p)\n\n    def TrainML(model_class, test_data):\n\n        X = train.drop(['sii'], axis=1)\n        y = train['sii']\n\n        SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n        train_S = []\n        test_S = []\n\n        oof_non_rounded = np.zeros(len(y), dtype=float) \n        oof_rounded = np.zeros(len(y), dtype=int) \n        test_preds = np.zeros((len(test_data), n_splits))\n\n        for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n            model = clone(model_class)\n            model.fit(X_train, y_train)\n\n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n\n            oof_non_rounded[test_idx] = y_val_pred\n            y_val_pred_rounded = y_val_pred.round(0).astype(int)\n            oof_rounded[test_idx] = y_val_pred_rounded\n\n            train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n            val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n            train_S.append(train_kappa)\n            test_S.append(val_kappa)\n\n            test_preds[:, fold] = model.predict(test_data)\n\n            print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            clear_output(wait=True)\n\n        print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n        print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n        KappaOPtimizer = minimize(evaluate_predictions,\n                                  x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                                  method='Nelder-Mead') # Nelder-Mead | # Powell\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n        tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n        print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n        submission = pd.DataFrame({\n            'id': sample['id'],\n            'sii': tpTuned\n        })\n\n        return submission","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.417864Z","iopub.execute_input":"2024-09-28T10:32:11.418225Z","iopub.status.idle":"2024-09-28T10:32:11.434226Z","shell.execute_reply.started":"2024-09-28T10:32:11.418177Z","shell.execute_reply":"2024-09-28T10:32:11.433361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_8' in ENSEMBLE_SOLUTIONS:\n\n    Params = {'learning_rate': 0.07975474666326936, 'max_depth': 10, 'num_leaves': 207, 'min_data_in_leaf': 41,\n                    'feature_fraction': 0.6385678848225935, 'bagging_fraction': 0.9042038292349021, 'bagging_freq': 6, \n                                'lambda_l1': 9.920617415343463, 'lambda_l2': 4.351491475117983} # LB : 0.452\n\n    Light = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1,n_estimators=200)\n    Submission = TrainML(Light,test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.435877Z","iopub.execute_input":"2024-09-28T10:32:11.436496Z","iopub.status.idle":"2024-09-28T10:32:11.447848Z","shell.execute_reply.started":"2024-09-28T10:32:11.436453Z","shell.execute_reply":"2024-09-28T10:32:11.447112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_8' in ENSEMBLE_SOLUTIONS:\n\n    Submission.to_csv('submission_8.csv', index=False)\n    Submission.head()\n    print(Submission['sii'].value_counts())","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.449150Z","iopub.execute_input":"2024-09-28T10:32:11.449481Z","iopub.status.idle":"2024-09-28T10:32:11.457804Z","shell.execute_reply.started":"2024-09-28T10:32:11.449448Z","shell.execute_reply":"2024-09-28T10:32:11.457103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_9' in ENSEMBLE_SOLUTIONS:\n    \n    import pandas as pd\n    import numpy as np\n    #model lightgbm回归模型,日志评估\n    from  lightgbm import LGBMClassifier,log_evaluation,early_stopping\n    #KFold是直接分成k折,StratifiedKFold还要考虑每种类别的占比\n    from sklearn.model_selection import StratifiedKFold\n    import warnings#避免一些可以忽略的报错\n    warnings.filterwarnings('ignore')#filterwarnings()方法是用于设置警告过滤器的方法，它可以控制警告信息的输出方式和级别。\n\n    #config\n    class Config():\n        seed=2024#随机种子\n        num_folds=10#K折交叉验证\n        TARGET_NAME ='sii'#标签\n    import random#提供了一些用于生成随机数的函数\n    #设置随机种子,保证模型可以复现\n    def seed_everything(seed):\n        np.random.seed(seed)#numpy的随机种子\n        random.seed(seed)#python内置的随机种子\n    seed_everything(Config.seed)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.459268Z","iopub.execute_input":"2024-09-28T10:32:11.459549Z","iopub.status.idle":"2024-09-28T10:32:11.470355Z","shell.execute_reply.started":"2024-09-28T10:32:11.459519Z","shell.execute_reply":"2024-09-28T10:32:11.469504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_9' in ENSEMBLE_SOLUTIONS:  \n    \n    train=pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\n    #target为缺失值,不要了\n    train=train[train['sii']==train['sii']]\n    test=pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\n    print(f\"len(train):{len(train)}\")\n    train.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.477274Z","iopub.execute_input":"2024-09-28T10:32:11.477525Z","iopub.status.idle":"2024-09-28T10:32:11.483555Z","shell.execute_reply.started":"2024-09-28T10:32:11.477497Z","shell.execute_reply":"2024-09-28T10:32:11.482761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_9' in ENSEMBLE_SOLUTIONS:  \n    \n    def FE(df):\n\n        print(\"Basic_Demos-Enroll_Season feature\")\n        col2count={'Spring': 734, 'Fall': 676, 'Winter': 676, 'Summer': 650}\n        for u,v in col2count.items():\n            df[f'Basic_Demos-Enroll_Season_{u}']=(df['Basic_Demos-Enroll_Season']==u).astype(np.int8)\n        df['Basic_Demos-Enroll_Season']=df['Basic_Demos-Enroll_Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"CGAS-Season feature\")\n        col2count={'Spring': 665, 'Fall': 583, 'Summer': 559, 'Winter': 535}\n        for u,v in col2count.items():\n            df[f'CGAS-Season_{u}']=(df['CGAS-Season']==u).astype(np.int8)\n        df['CGAS-Season']=df['CGAS-Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"Physical-Season feature\")\n        col2count={'Spring': 709, 'Fall': 650, 'Winter': 634, 'Summer': 602}\n        for u,v in col2count.items():\n            df[f'Physical-Season_{u}']=(df['Physical-Season']==u).astype(np.int8)\n        df['Physical-Season']=df['Physical-Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"Fitness_Endurance-Season feature\")\n        col2count={'Spring': 377, 'Winter': 331, 'Fall': 316, 'Summer': 236}\n        for u,v in col2count.items():\n            df[f'Fitness_Endurance-Season_{u}']=(df['Fitness_Endurance-Season']==u).astype(np.int8)\n        df['Fitness_Endurance-Season']=df['Fitness_Endurance-Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"FGC-Season feature\")\n        col2count={'Spring': 771, 'Summer': 659, 'Fall': 633, 'Winter': 584}\n        for u,v in col2count.items():\n            df[f'FGC-Season_{u}']=(df['FGC-Season']==u).astype(np.int8)\n        df['FGC-Season']=df['FGC-Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"BIA-Season feature\")\n        col2count={'Summer': 585, 'Fall': 511, 'Spring': 405, 'Winter': 343}\n        for u,v in col2count.items():\n            df[f'BIA-Season_{u}']=(df['BIA-Season']==u).astype(np.int8)\n        df['BIA-Season']=df['BIA-Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"PAQ_A-Season feature\")\n        col2count={'Winter': 98, 'Summer': 97, 'Spring': 90, 'Fall': 78}   \n        for u,v in col2count.items():\n            df[f'PAQ_A-Season_{u}']=(df['PAQ_A-Season']==u).astype(np.int8)\n        df['PAQ_A-Season']=df['PAQ_A-Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"PAQ_C-Season feature\")\n        col2count={'Spring': 405, 'Winter': 385, 'Summer': 342, 'Fall': 308}\n        for u,v in col2count.items():\n            df[f'PAQ_C-Season_{u}']=(df['PAQ_C-Season']==u).astype(np.int8)\n        df['PAQ_C-Season']=df['PAQ_C-Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"SDS-Season feature\")\n        col2count={'Spring': 692, 'Winter': 629, 'Fall': 605, 'Summer': 601}\n        for u,v in col2count.items():\n            df[f'SDS-Season_{u}']=(df['SDS-Season']==u).astype(np.int8)\n        df['SDS-Season']=df['SDS-Season'].apply(lambda x:col2count.get(x,0))\n\n        print(\"PreInt_EduHx-Season feature\")\n        col2count={'Spring': 728, 'Fall': 684, 'Winter': 655, 'Summer': 652}\n        for u,v in col2count.items():\n            df[f'PreInt_EduHx-Season_{u}']=(df['PreInt_EduHx-Season']==u).astype(np.int8)\n        df['PreInt_EduHx-Season']=df['PreInt_EduHx-Season'].apply(lambda x:col2count.get(x,0)) \n\n        test_lack_cols=['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total']\n        df.drop(['id',#id只有1个,没什么实际含义\n        ]+test_lack_cols,axis=1,inplace=True,errors='ignore')\n        print(\"-\"*30)\n        return df\n    train=FE(train)\n    test=FE(test)\n    train.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.487392Z","iopub.execute_input":"2024-09-28T10:32:11.487725Z","iopub.status.idle":"2024-09-28T10:32:11.509870Z","shell.execute_reply.started":"2024-09-28T10:32:11.487693Z","shell.execute_reply":"2024-09-28T10:32:11.509025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_9' in ENSEMBLE_SOLUTIONS:   \n    \n    #weighted kappa的手写\n    def metric(y_true,y_pred):\n        #y_true是label,y_pred是我的预测结果\n        #y_true为0,1,2,……,i,所以类别数为i+1\n        N=int(np.max(y_true)+1)\n        #混淆矩阵 真实值为i预测为j的count\n        O=np.zeros((N,N))\n        for i in range(len(y_true)):\n            O[y_true[i]][y_pred[i]]+=1\n        #W权重系数\n        W=np.zeros((N,N))\n        for i in range(N):\n            for j in range(N):\n                W[i][j]=(i-j)**2/(N-1)**2\n        #E\n        true_count=np.zeros(N)\n        pred_count=np.zeros(N)\n        for i in range(len(y_true)):\n            true_count[y_true[i]]+=1\n            pred_count[y_pred[i]]+=1\n\n        E=np.zeros((N,N))\n        for i in range(len(true_count)):\n            for j in range(len(pred_count)):\n                E[i][j]=true_count[i]*pred_count[j]\n\n        O=O/O.sum()\n        E=E/E.sum()\n\n        weighted_kappa=1-np.sum(W*O)/np.sum(W*E)\n\n        return weighted_kappa\n\n\n    choose_cols=[col for col in test.columns if train[col].dtype!=object]\n    print(f\"len(choose_cols):{len(choose_cols)}\")\n\n    def fit_and_predict(train_feats=train,test_feats=test,model=None,num_folds=10,name='lgb'):\n        X=train_feats[choose_cols].copy()\n        y=train_feats[Config.TARGET_NAME].copy()\n        oof_pred=np.zeros((len(X)))\n        test_X=test_feats[choose_cols].copy()\n        test_pred_pro=np.zeros((num_folds,len(test_X)))\n\n        #k折交叉验证\n        skf = StratifiedKFold(n_splits=num_folds,shuffle=True,random_state=Config.seed)\n        for fold, (train_index, valid_index) in (enumerate(skf.split(X,y))):\n            print(f\"name {name},fold:{fold}\")\n\n            X_train, X_valid = X.iloc[train_index].reset_index(drop=True), X.iloc[valid_index].reset_index(drop=True)\n            y_train, y_valid = y.iloc[train_index].reset_index(drop=True), y.iloc[valid_index].reset_index(drop=True)\n\n            #训练前的transform\n            y_train=(y_train!=0).astype(np.int8)\n            y_valid=(y_valid!=0).astype(np.int8)\n\n            model.fit(X_train,y_train,eval_set=[(X_valid, y_valid)],\n                              callbacks=[log_evaluation(100)],\n                             )\n\n            oof_pred[valid_index]=model.predict_proba(X_valid)[:,1]\n            test_pred_pro[fold]=model.predict_proba(test_X)[:,1]\n\n            #1000个迭代器,1%的特征\n            importances=model.feature_importances_\n            columns=X.columns\n            drop_cols=[]\n            for i in range(len(columns)):\n                if importances[i]<10:\n                    drop_cols.append(columns[i])\n            print(f\"drop_cols={drop_cols}\")\n\n        test_pred_pro=test_pred_pro.mean(axis=0)\n\n        #测试集和oof按照占比label为[0,1,2,3]   [0,0.58,0.85,0.99,1]\n        oof_pred_sorted=sorted(oof_pred)\n        margin=[0,oof_pred_sorted[int(len(oof_pred_sorted)*0.58)],\n         oof_pred_sorted[int(len(oof_pred_sorted)*0.85)],oof_pred_sorted[int(len(oof_pred_sorted)*0.99)],1]\n        for i in range(len(margin)-1):\n            oof_index= np.where((oof_pred >= margin[i]) & (oof_pred < margin[i+1]))[0]\n            oof_pred[oof_index]=i\n            test_index= np.where((test_pred_pro >= margin[i]) & (test_pred_pro < margin[i+1]))[0]\n            test_pred_pro[test_index]=i\n\n        print(f\"weighted kappa:{metric(y.values.reshape(-1).astype(np.int8),oof_pred.astype(np.int8))}\")\n\n        return oof_pred,test_pred_pro\n\n    lgb_params={\n            \"boosting_type\": \"gbdt\",'objective':'binary',\"metric\": \"auc\",\n            'random_state': 2024, 'n_estimators': 512, \n            'learning_rate': 0.07975474666326936, 'max_depth': 10, 'num_leaves': 207,\n            'min_data_in_leaf': 41,'feature_fraction': 0.6385678848225935, \n            'bagging_fraction': 0.9042038292349021, 'bagging_freq': 6, \n            'lambda_l1': 9.920617415343463, 'lambda_l2': 4.351491475117983,\n            'reg_alpha': 0.006329813118558037, 'reg_lambda': 0.22366541275310856,\n            'colsample_bytree': 0.9045121369263609, 'subsample': 0.6560250299728694,\n            'min_child_samples': 16,\"verbose\": -1, \n            #'device':'gpu','gpu_use_dp':True,#这行GPU环境的参数,想在CPU环境下运行注释这行代码\n    }\n\n    lgb_oof_pred_pro,lgb_test_pro=fit_and_predict(model=LGBMClassifier(**lgb_params),num_folds=Config.num_folds,name='lgb')\n    print(f\"lgb_test_pro[:10]:{lgb_test_pro[:10]}\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.511831Z","iopub.execute_input":"2024-09-28T10:32:11.512221Z","iopub.status.idle":"2024-09-28T10:32:11.537359Z","shell.execute_reply.started":"2024-09-28T10:32:11.512177Z","shell.execute_reply":"2024-09-28T10:32:11.536493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_9' in ENSEMBLE_SOLUTIONS:   \n    \n    submission=pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv\")\n    submission[Config.TARGET_NAME]=lgb_test_pro\n    submission.to_csv(\"submission_9.csv\",index=None)\n    submission.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.538255Z","iopub.execute_input":"2024-09-28T10:32:11.538503Z","iopub.status.idle":"2024-09-28T10:32:11.550213Z","shell.execute_reply.started":"2024-09-28T10:32:11.538474Z","shell.execute_reply":"2024-09-28T10:32:11.549463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_10' in ENSEMBLE_SOLUTIONS:    \n    \n    import numpy as np\n    import polars as pl\n    import pandas as pd\n    from sklearn.base import clone\n    from copy import deepcopy\n    import optuna\n    from scipy.optimize import minimize\n\n    import re\n    from colorama import Fore, Style\n\n    from tqdm import tqdm\n    from IPython.display import clear_output\n\n    import warnings\n    warnings.filterwarnings('ignore')\n    pd.options.display.max_columns = None\n\n    import lightgbm as lgb\n    from catboost import CatBoostRegressor, CatBoostClassifier\n    from xgboost import XGBRegressor\n    from sklearn.ensemble import VotingRegressor\n    from sklearn.model_selection import *\n    from sklearn.metrics import *\n    import os\n\n    SEED = 42\n    n_splits = 5","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.552042Z","iopub.execute_input":"2024-09-28T10:32:11.552352Z","iopub.status.idle":"2024-09-28T10:32:11.559649Z","shell.execute_reply.started":"2024-09-28T10:32:11.552320Z","shell.execute_reply":"2024-09-28T10:32:11.558936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_10' in ENSEMBLE_SOLUTIONS:\n\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n    def load_time_series(dirname) -> pd.DataFrame:\n        ids = os.listdir(dirname)\n        indexes = []\n        stats = []\n        for idname in tqdm(ids):\n            df = pd.read_parquet(os.path.join(dirname, idname, 'part-0.parquet'))\n            df.drop('step', axis=1, inplace=True)\n            stats.append(df.describe().values.reshape(-1))\n            indexes.append(idname.split('=')[1])\n        df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n        df['id'] = indexes\n        return df\n\n    train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n    test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n    time_series_cols = train_ts.columns.tolist()\n    time_series_cols.remove(\"id\")\n\n    train = pd.merge(train, train_ts, how=\"left\", on='id')\n    test = pd.merge(test, test_ts, how=\"left\", on='id')\n\n    train = train.drop('id',axis=1)\n    test = test.drop('id',axis=1)\n\n    featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\n\n    featuresCols += time_series_cols\n\n    train = train[featuresCols]\n    train = train.dropna(subset='sii')\n\n    cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\n    def update(df):\n        global cat_c\n        for c in cat_c : \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n\n        return df\n\n    train = update(train)\n    test = update(test)\n\n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n\n    for col in cat_c:\n\n        mapping = create_mapping(col, train)\n        mappingTe = create_mapping(col, test)\n\n        train[col] = train[col].replace(mapping).astype(int)\n        test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.560800Z","iopub.execute_input":"2024-09-28T10:32:11.561120Z","iopub.status.idle":"2024-09-28T10:32:11.577421Z","shell.execute_reply.started":"2024-09-28T10:32:11.561085Z","shell.execute_reply":"2024-09-28T10:32:11.576633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_10' in ENSEMBLE_SOLUTIONS:\n\n    train.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.578497Z","iopub.execute_input":"2024-09-28T10:32:11.579137Z","iopub.status.idle":"2024-09-28T10:32:11.590874Z","shell.execute_reply.started":"2024-09-28T10:32:11.579092Z","shell.execute_reply":"2024-09-28T10:32:11.589928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_10' in ENSEMBLE_SOLUTIONS:\n\n    test.head()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.593027Z","iopub.execute_input":"2024-09-28T10:32:11.593329Z","iopub.status.idle":"2024-09-28T10:32:11.599777Z","shell.execute_reply.started":"2024-09-28T10:32:11.593298Z","shell.execute_reply":"2024-09-28T10:32:11.599019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_10' in ENSEMBLE_SOLUTIONS:\n\n    def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n    def threshold_Rounder(oof_non_rounded, thresholds):\n        return np.where(oof_non_rounded < thresholds[0], 0,\n                        np.where(oof_non_rounded < thresholds[1], 1,\n                                 np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n    def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n        rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_p)\n\n    def TrainML(model_class, test_data):\n\n        X = train.drop(['sii'], axis=1)\n        y = train['sii']\n\n        SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n        train_S = []\n        test_S = []\n\n        oof_non_rounded = np.zeros(len(y), dtype=float) \n        oof_rounded = np.zeros(len(y), dtype=int) \n        test_preds = np.zeros((len(test_data), n_splits))\n\n        for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n            model = clone(model_class)\n            model.fit(X_train, y_train)\n\n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n\n            oof_non_rounded[test_idx] = y_val_pred\n            y_val_pred_rounded = y_val_pred.round(0).astype(int)\n            oof_rounded[test_idx] = y_val_pred_rounded\n\n            train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n            val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n            train_S.append(train_kappa)\n            test_S.append(val_kappa)\n\n            test_preds[:, fold] = model.predict(test_data)\n\n            print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            clear_output(wait=True)\n\n        print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n        print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n        KappaOPtimizer = minimize(evaluate_predictions,\n                                  x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                                  method='Nelder-Mead') # Nelder-Mead | # Powell\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n        tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n        print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n        submission = pd.DataFrame({\n            'id': sample['id'],\n            'sii': tpTuned\n        })\n\n        return submission","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.601227Z","iopub.execute_input":"2024-09-28T10:32:11.601519Z","iopub.status.idle":"2024-09-28T10:32:11.618010Z","shell.execute_reply.started":"2024-09-28T10:32:11.601487Z","shell.execute_reply":"2024-09-28T10:32:11.617127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_10' in ENSEMBLE_SOLUTIONS:\n\n    Params = {'learning_rate': 0.07975474666326936, 'max_depth': 10, 'num_leaves': 207, 'min_data_in_leaf': 41,\n                    'feature_fraction': 0.6385678848225935, 'bagging_fraction': 0.9042038292349021, 'bagging_freq': 6, \n                                'lambda_l1': 9.920617415343463, 'lambda_l2': 4.351491475117983} # LB : 0.452\n\n    Light = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1,n_estimators=200)\n    Submission = TrainML(Light,test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.619105Z","iopub.execute_input":"2024-09-28T10:32:11.619660Z","iopub.status.idle":"2024-09-28T10:32:11.631992Z","shell.execute_reply.started":"2024-09-28T10:32:11.619618Z","shell.execute_reply":"2024-09-28T10:32:11.631122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_10' in ENSEMBLE_SOLUTIONS:\n\n    Submission.to_csv('submission_10.csv', index=False)\n    Submission.head()\n    print(Submission['sii'].value_counts())","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.633109Z","iopub.execute_input":"2024-09-28T10:32:11.633385Z","iopub.status.idle":"2024-09-28T10:32:11.642634Z","shell.execute_reply.started":"2024-09-28T10:32:11.633344Z","shell.execute_reply":"2024-09-28T10:32:11.641778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_11' in ENSEMBLE_SOLUTIONS: pass","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.644923Z","iopub.execute_input":"2024-09-28T10:32:11.645278Z","iopub.status.idle":"2024-09-28T10:32:11.654116Z","shell.execute_reply.started":"2024-09-28T10:32:11.645237Z","shell.execute_reply":"2024-09-28T10:32:11.653233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture \n\n!pip install -q lightgbm==4.5.0 --no-index --find-links=/kaggle/input/cmi2024-packages-v1/MLPackages\n!pip install -q polars==1.7.1 --no-index --find-links=/kaggle/input/cmi2024-packages-v1/Polars171","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:11.655223Z","iopub.execute_input":"2024-09-28T10:32:11.655539Z","iopub.status.idle":"2024-09-28T10:32:15.693566Z","shell.execute_reply.started":"2024-09-28T10:32:11.655492Z","shell.execute_reply":"2024-09-28T10:32:15.692217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"\\n---> Commencing imports\")\n\nfrom gc import collect\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom IPython.display import display_html, clear_output\nclear_output()\nimport os, sys, logging, re, joblib, ctypes, shutil\nfrom copy import deepcopy\n\n# General library imports:-\nfrom os import path, walk, getpid\nfrom psutil import Process\nfrom collections import Counter\nfrom itertools import product\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport pandas as pd, numpy as np\nfrom scipy.optimize import minimize\nfrom numpy.typing import ArrayLike, NDArray\nimport polars as pl\nimport polars.selectors as cs\nfrom polars.testing import assert_frame_equal\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import Fore, Style, init\nfrom tqdm.notebook import tqdm","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:15.696286Z","iopub.execute_input":"2024-09-28T10:32:15.696633Z","iopub.status.idle":"2024-09-28T10:32:15.704433Z","shell.execute_reply.started":"2024-09-28T10:32:15.696590Z","shell.execute_reply":"2024-09-28T10:32:15.703217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Importing model and pipeline specifics:-\nfrom category_encoders import OrdinalEncoder, OneHotEncoder\n\n# Pipeline specifics:-\nfrom sklearn.preprocessing import (RobustScaler,\n                                   MinMaxScaler,\n                                   StandardScaler,\n                                   FunctionTransformer as FT,\n                                   PowerTransformer,\n                                  )\nfrom sklearn.impute import SimpleImputer as SI\nfrom sklearn.model_selection import (RepeatedStratifiedKFold as RSKF,\n                                     StratifiedKFold as SKF,\n                                     StratifiedGroupKFold as SGKF,\n                                     KFold,\n                                     GroupKFold as GKF,\n                                     RepeatedKFold as RKF,\n                                     PredefinedSplit as PDS,\n                                     cross_val_score,\n                                     cross_val_predict,\n                                    )\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.feature_selection import VarianceThreshold as VT\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.base import (BaseEstimator, TransformerMixin, RegressorMixin, clone)\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import (mean_squared_error as mse, \n                             cohen_kappa_score,\n                             ConfusionMatrixDisplay,\n                             confusion_matrix,\n                            )\n\n# Importing model packages\nimport xgboost as xgb, lightgbm as lgb\nfrom xgboost import QuantileDMatrix, XGBRegressor as XGBR\nfrom lightgbm import log_evaluation, early_stopping, LGBMRegressor as LGBMR\nfrom catboost import CatBoostRegressor as CBR, Pool\n\n# Importing Ensemble and tuning packages\nimport optuna\nfrom optuna import Trial, trial, create_study\nfrom optuna.pruners import HyperbandPruner\nfrom optuna.samplers import TPESampler, CmaEsSampler\noptuna.logging.disable_default_handler()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:15.706025Z","iopub.execute_input":"2024-09-28T10:32:15.706300Z","iopub.status.idle":"2024-09-28T10:32:15.722034Z","shell.execute_reply.started":"2024-09-28T10:32:15.706268Z","shell.execute_reply":"2024-09-28T10:32:15.721078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Setting rc parameters in seaborn for plots and graphs-\nsns.set({\"axes.facecolor\"       : \"#ffffff\",\n         \"figure.facecolor\"     : \"#ffffff\",\n         \"axes.edgecolor\"       : \"#000000\",\n         \"grid.color\"           : \"#ffffff\",\n         \"font.family\"          : ['Cambria'],\n         \"axes.labelcolor\"      : \"#000000\",\n         \"xtick.color\"          : \"#000000\",\n         \"ytick.color\"          : \"#000000\",\n         \"grid.linewidth\"       : 0.75,\n         \"grid.linestyle\"       : \"--\",\n         \"axes.titlecolor\"      : '#0099e6',\n         'axes.titlesize'       : 8.5,\n         'axes.labelweight'     : \"bold\",\n         'legend.fontsize'      : 7.0,\n         'legend.title_fontsize': 7.0,\n         'font.size'            : 7.5,\n         'xtick.labelsize'      : 7.5,\n         'ytick.labelsize'      : 7.5,\n        }\n       )\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n \n# Checking package versions\nimport xgboost as xgb, lightgbm as lgb, catboost as cb, sklearn as sk\nprint(f\"---> XGBoost = {xgb.__version__} | LightGBM = {lgb.__version__} | Catboost = {cb.__version__}\")\nprint(f\"---> Sklearn = {sk.__version__}| Pandas = {pd.__version__} | Polars = {pl.__version__}\")\ncollect()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:15.723319Z","iopub.execute_input":"2024-09-28T10:32:15.723616Z","iopub.status.idle":"2024-09-28T10:32:15.735514Z","shell.execute_reply.started":"2024-09-28T10:32:15.723585Z","shell.execute_reply":"2024-09-28T10:32:15.734518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nclass MyLogger:\n    \"\"\"\n    This class helps to suppress logs in lightgbm and Optuna\n    Source - https://github.com/microsoft/LightGBM/issues/6014\n    \"\"\"\n\n    def init(self, logging_lbl: str):\n        self.logger = logging.getLogger(logging_lbl)\n        self.logger.setLevel(logging.ERROR)\n\n    def info(self, message):\n        pass\n\n    def warning(self, message):\n        pass\n\n    def error(self, message):\n        self.logger.error(message)\n        \n        \n# Customizing logging for XGBoost\nfor handler in logging.root.handlers[:]:\n    logging.root.removeHandler(handler)\n\nlogger = logging.getLogger(__name__)\nlogger.setLevel(logging.ERROR)\nformatter = logging.Formatter('%(asctime)s | %(levelname)s | %(message)s')\n\nstdout_handler = logging.StreamHandler(sys.stdout)\nstdout_handler.setLevel(logging.INFO)\nstdout_handler.setFormatter(formatter)\n\nfile_handler = logging.FileHandler(f'xgb_optimize.log')\nfile_handler.setLevel(logging.ERROR)\nfile_handler.setFormatter(formatter)\n\nlogger.addHandler(file_handler)\nlogger.addHandler(stdout_handler)\n\nclass XGBLogging(xgb.callback.TrainingCallback):\n    \"\"\"\n    This class designs the custom logging for XGboost \n    This is to be used inside the XGboost callback\n    \"\"\"\n\n    def __init__(self, epoch_log_interval=100):\n        self.epoch_log_interval = epoch_log_interval\n\n    def after_iteration(\n        self, model, epoch:int, evals_log:xgb.callback.TrainingCallback.EvalsLog\n    ):\n        if self.epoch_log_interval <= 0:\n            pass\n        \n        elif (epoch %  self.epoch_log_interval == 0):\n            for data, metric in evals_log.items():\n                for metric_name, log in metric.items():\n                    score = log[-1][0] if isinstance(log[-1], tuple) else log[-1]\n                    logger.info(f\"XGBLogging epoch {epoch} dataset {data} {metric_name} {score}\")\n\n        return False","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:15.737047Z","iopub.execute_input":"2024-09-28T10:32:15.737405Z","iopub.status.idle":"2024-09-28T10:32:15.750829Z","shell.execute_reply.started":"2024-09-28T10:32:15.737346Z","shell.execute_reply":"2024-09-28T10:32:15.749893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nexec(open('myimports.py','r').read())\nprint()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:15.752034Z","iopub.execute_input":"2024-09-28T10:32:15.752399Z","iopub.status.idle":"2024-09-28T10:32:22.240092Z","shell.execute_reply.started":"2024-09-28T10:32:15.752362Z","shell.execute_reply":"2024-09-28T10:32:22.239180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\n# Configuration class:-\nclass CFG:\n    \"\"\"\n    Configuration class for parameters and CV strategy for tuning and training\n    Some parameters may be unused here as this is a general configuration class\n    \"\"\";\n\n    # Data preparation:-\n    version_nb  = 2\n    model_id    = \"V2_2\"\n    model_label = \"ML\"\n\n    test_req           = False\n    test_sample_frac   = 1000\n\n    gpu_switch         = \"OFF\"\n    state              = 42\n    target             = f\"sii\"\n    grouper            = f\"sii\"\n\n    ip_path            = f\"/kaggle/input/child-mind-institute-problematic-internet-use\"\n    op_path            = f\"/kaggle/working\"\n\n    # Model Training:-\n    pstprcs_oof        = False\n    pstprcs_train      = False\n    pstprcs_test       = False\n    ML                 = True\n    test_preds_req     = True\n\n    pseudo_lbl_req     = False\n    pseudolbl_up       = 0.975\n    pseudolbl_low      = 0.00\n\n    n_splits           = 3 if test_req == \"Y\" else 5\n    n_repeats          = 1\n    nbrnd_erly_stp     = 40\n    mdlcv_mthd         = 'RSKF'\n\n    # Ensemble:-\n    ensemble_req       = True\n    optuna_req         = True\n    metric_obj         = 'maximize'\n    ntrials            = 10 if test_req == \"Y\" else 300\n\n    # Global variables for plotting:-\n    grid_specs = {'visible'  : True,\n                  'which'    : 'both',\n                  'linestyle': '--',\n                  'color'    : 'lightgrey',\n                  'linewidth': 0.75\n                 }\n\n    title_specs = {'fontsize'   : 9,\n                   'fontweight' : 'bold',\n                   'color'      : '#992600',\n                  }\n    \ncv_selector = \\\n{\n \"RKF\"   : RKF(n_splits = CFG.n_splits, n_repeats= CFG.n_repeats, random_state= CFG.state),\n \"RSKF\"  : RSKF(n_splits = CFG.n_splits, n_repeats= CFG.n_repeats, random_state= CFG.state),\n \"SKF\"   : SKF(n_splits = CFG.n_splits, shuffle = True, random_state= CFG.state),\n \"KF\"    : KFold(n_splits = CFG.n_splits, shuffle = True, random_state= CFG.state),\n}\n\nPrintColor(f\"\\n---> Configuration done!\\n\")\ncollect()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:22.244067Z","iopub.execute_input":"2024-09-28T10:32:22.245220Z","iopub.status.idle":"2024-09-28T10:32:22.252483Z","shell.execute_reply.started":"2024-09-28T10:32:22.245173Z","shell.execute_reply":"2024-09-28T10:32:22.251363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass Utils:\n    \"\"\"\n    This class creates and uses several utility methods to be used across the code\n    \"\"\";\n\n    def __init__(self):\n        pass\n\n    def ScoreMetric(self, ytrue, ypred)-> float:\n        \"\"\"\n        This method calculates the metric for the competition\n        Inputs- ytrue, ypred:- input truth and predictions\n        Output- float:- competition metric\n        \"\"\"\n        myscore = \\\n        cohen_kappa_score(\n            np.uint8(np.round(ytrue)), \n            np.uint8(np.round(ypred)), \n            weights = \"quadratic\"\n        )\n        return myscore\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\"\n\n        collect();\n        libc.malloc_trim(0)\n        pid        = getpid()\n        py         = Process(pid)\n        memory_use = py.memory_info()[0] / 2. ** 30\n        return f\"\\nRAM usage = {memory_use :.4} GB\"\n\n    def DisplayAdjTbl(self, *args):\n        \"\"\"\n        This function displays pandas tables in an adjacent manner, sourced from the below link-\n        https://stackoverflow.com/questions/38783027/jupyter-notebook-display-two-pandas-tables-side-by-side\n        \"\"\"\n\n        html_str = ''\n        for df in args:\n            html_str += df.to_html()\n        display_html(html_str.replace('table','table style=\"display:inline\"'),raw=True)\n        collect()\n\n    def DisplayScores(\n        self, Scores: pd.DataFrame, TrainScores: pd.DataFrame, methods: list\n    ):\n        \"This method displays the scores and their means\"\n\n        args = \\\n        [Scores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Blues\", subset = methods + [\"Ensemble\"]).\\\n         set_caption(f\"\\nOOF scores across methods and folds\\n\"),\n\n         TrainScores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Pastel2\", subset = methods).\\\n         set_caption(f\"\\nTrain scores across methods and folds\\n\")\n        ];\n\n        PrintColor(f\"\\n\\n\\n---> OOF score across all methods and folds\\n\",\n                   color = Fore.LIGHTMAGENTA_EX\n                   )\n        self.DisplayAdjTbl(*args)\n\n        print('\\n')\n        display(Scores.mean().to_frame().\\\n                transpose().\\\n                style.format(precision = 5).\\\n                background_gradient(cmap = \"mako\", axis=1,\n                                    subset = Scores.columns\n                                   ).\\\n                set_caption(f\"\\nOOF mean scores across methods and folds\\n\")\n               )\n\n\nutils = Utils()\ncollect()\nprint()","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:22.253906Z","iopub.execute_input":"2024-09-28T10:32:22.254435Z","iopub.status.idle":"2024-09-28T10:32:22.270331Z","shell.execute_reply.started":"2024-09-28T10:32:22.254381Z","shell.execute_reply":"2024-09-28T10:32:22.269427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass Preprocessor:\n    \"This class organizes the preprocessing steps for the train-test data into a single code block\"\n    \n    def __init__(\n        self, cat_imp_val : str= \"missing\", ip_path: str = CFG.ip_path\n    ):\n        self.cat_imp_val = cat_imp_val\n        self.ip_path     = ip_path\n        \n    def make_pqfile_cols(\n        self, verbose: bool = False, label: str = \"Train\"\n    )->pl.DataFrame:\n        \"This method collates the id level parquet files and creates the aggregation columns in a polars dataframe\"\n\n        cols = [\"X\", \"Y\", \"Z\", \"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n        \n        ip_path   = os.path.join(self.ip_path, f\"series_{label.lower()}.parquet\")\n        all_files = os.listdir(ip_path)\n\n        for file_nb, file in tqdm(enumerate(all_files)):\n            df = \\\n            pl.scan_parquet(\n                os.path.join(ip_path, file, f\"part-0.parquet\")\n            ).select(pl.col(cols)).\\\n            collect().\\\n            describe(\n                percentiles = np.arange(0.05, 0.95, 0.10)\n            ).\\\n            filter(~pl.col(\"statistic\").is_in([\"count\", \"null_count\"])).\\\n            unpivot(index = \"statistic\").\\\n            with_columns(\n                pl.concat_str([pl.col(\"variable\"), pl.col(\"statistic\")],separator = \"_\",).alias(\"myvar\")\n            ).\\\n            with_columns(pl.col(\"myvar\").str.replace(r\"\\%\", \"\")).\\\n            select([\"myvar\", \"value\"]).\\\n            transpose(column_names = \"myvar\").\\\n            select(pl.all().shrink_dtype()).\\\n            with_columns(\n                pl.Series(\"id\", np.array(re.sub(\"id=\", \"\", file)))\n            )\n\n            if file_nb == 0:\n                op_df = df.clone()\n            elif file_nb > 0:\n                op_df = pl.concat([op_df, df], how = \"vertical_relaxed\")\n\n                if verbose:\n                    print(f\"---> Shapes = {op_df.shape}\")\n                else:\n                    pass\n            del df\n\n        PrintColor(f\"---> {label} - shape = {op_df.shape}\", color = Fore.CYAN)\n        return op_df\n\n    def pp_data(\n        self, df: pl.DataFrame, label: str = \"Train\", cat_cols: list = [], \n    ):\n        \"This method preprocesses the train-test data with requisite steps\"\n        \n        PrintColor(f\"\\n --- Data Processing - {label} --- \\n\")\n        PrintColor(f\"---> Shape = {df.shape} - memory usage {df.estimated_size('mb') :.3f} MB\", \n                   color = Fore.CYAN\n                  )\n        \n        if label == \"Train\":\n            cat_cols = df.select(cs.string().exclude(\"id\")).columns\n        else:\n            pass\n        \n        df    = df.with_columns(pl.col(cat_cols).fill_null(self.cat_imp_val).cast(pl.Categorical))\n        op_df = self.make_pqfile_cols(label = label)\n        df    = df.join(op_df, how = \"left\", on = \"id\")\n        df    = df.select(pl.all().shrink_dtype())\n        del op_df\n        \n        PrintColor(f\"---> Shape = {df.shape} - memory usage {df.estimated_size('mb') :.3f} MB\", \n                   color = Fore.CYAN\n                  )\n        return df, cat_cols\n        ","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-09-28T10:32:22.271419Z","iopub.execute_input":"2024-09-28T10:32:22.271736Z","iopub.status.idle":"2024-09-28T10:32:22.287077Z","shell.execute_reply.started":"2024-09-28T10:32:22.271703Z","shell.execute_reply":"2024-09-28T10:32:22.286183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nexec(open('training.py','r').read())\ntrain  = pl.read_csv(os.path.join(CFG.ip_path, \"train.csv\")).drop(\"PCIAT-Season\", strict = False)\ntest   = pl.read_csv(os.path.join(CFG.ip_path, \"test.csv\"))\nsub_fl = pl.read_csv(os.path.join(CFG.ip_path, \"sample_submission.csv\"))\n\npp    = Preprocessor()\ntrain, cat_cols = pp.pp_data(train, \"Train\")\ntest, _  = pp.pp_data(test, \"Test\", cat_cols)\nsel_cols = test.drop(\"id\", strict = False).columns\n\nprint()\n_ = utils.CleanMemory()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-09-28T10:32:22.288230Z","iopub.execute_input":"2024-09-28T10:32:22.288510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass ModelTrainer:\n    \"This class trains the provided model on the train-test data and returns the predictions and fitted models\"\n\n    def __init__(\n        self,\n        es_req         : bool  = False,\n        es             : int   = 100,\n        target         : str   = CFG.target,\n        metric_lbl     : str   = \"kappa\",\n        drop_cols      : list  = [\"Source\", \"id\", \"Id\", \"Label\", CFG.target, \"fold_nb\"],\n    ):\n        \"\"\"\n        Key parameters-\n        es_iter - early stopping rounds for boosted trees\n        \"\"\"\n        \n        drop_cols = list(set(drop_cols + [target]))\n        \n        self.es_req         = es_req\n        self.es_iter        = es\n        self.target         = target\n        self.drop_cols      = drop_cols\n        self.metric_lbl     = metric_lbl\n        \n    def ScoreMetric(self, ytrue, ypred)->float:\n        \"\"\"\n        This is the metric function for the competition scoring\n        \"\"\"\n        if self.metric_lbl == \"rmse\":\n            return mse(ytrue, ypred, squared = False)\n        \n        elif self.metric_lbl == \"kappa\":\n            myscore = \\\n            cohen_kappa_score(\n                np.uint8(np.around(ytrue,0)),\n                np.uint8(np.around(ypred,0)),\n                weights = \"quadratic\",\n            )\n            return myscore\n\n    def PlotFtreImp(\n        self, ftreimp: pd.Series, method: str,\n        ntop: int = 50,\n        title_specs: dict = CFG.title_specs,\n        **params,\n    ):\n        \"This function plots the feature importances for the model provided\"\n\n        print()\n        fig, ax = plt.subplots(1, 1, figsize = (25, 7.5))\n\n        ftreimp.sort_values(ascending = False).\\\n        head(ntop).\\\n        plot.bar(ax = ax, color = \"blue\")\n        ax.set_title(f\"Feature Importances - {method}\", **title_specs)\n\n        plt.tight_layout()\n        plt.show()\n        print()\n\n    def PostProcessPreds(self, ypred):\n        \"This method post-processes predictions optionally\"\n        return np.clip(ypred, a_min = 0, a_max = np.inf)\n\n    def MakeOfflineModel(\n        self, X, y, ygrp, Xtest, mdl, method,\n        test_preds_req   : bool = True,\n        ftreimp_plot_req : bool = True,\n        ntop             : int  = 50,\n        **params,\n    ):\n        \"\"\"\n        This function trains the provided model on the dataset and cross-validates appropriately\n\n        Inputs-\n        X, y, ygrp       - training data components\n        Xtest            - test data\n        model            - model object for training\n        method           - model method label\n        test_preds_req   - boolean flag to extract test set predictions\n        ftreimp_plot_req - boolean flag to plot tree feature importances\n        ntop             - top n features for feature importances plot\n\n        Returns-\n        oof_preds, test_preds - prediction arrays\n        fitted_models         - fitted model list for test set\n        ftreimp               - feature importances across selected features\n        mdl_best_iter         - model average best iteration across folds\n        \"\"\"\n\n        oof_preds     = np.zeros(len(X))\n        test_preds    = []\n        mdl_best_iter = []\n        ftreimp       = 0\n        \n        scores, tr_scores, fitted_models = [], [], []\n        cv = PDS(ygrp)\n        n_splits = ygrp.nunique()\n\n        for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n            Xtr   = X.iloc[train_idx]\n            Xdev  = X.iloc[dev_idx]\n            ytr   = y.iloc[train_idx]\n            ydev  = y.iloc[dev_idx]\n            model = clone(mdl)\n\n            if \"CB\" in method and self.es_req == True:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose = 0,\n                          early_stopping_rounds = self.es_iter,\n                          )\n                best_iter = model.get_best_iteration()\n\n            elif \"LGB\" in method and self.es_req == True:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          callbacks = [log_evaluation(0),\n                                       early_stopping(stopping_rounds = self.es_iter, verbose = False,),\n                                       ],\n                          eval_metric = MakeEvalMetric,\n                          )\n                best_iter = model.best_iteration_\n\n            elif \"XGB\" in method and self.es_req == True:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose  = 0,\n                          )\n                best_iter = model.best_iteration\n\n            else:\n                model.fit(Xtr, ytr)\n                best_iter = -1\n\n            fitted_models.append(model)\n\n            try:\n                ftreimp += model.feature_importances_\n            except:\n                pass\n\n            dev_preds = self.PostProcessPreds(model.predict(Xdev))\n            oof_preds[Xdev.index] = dev_preds\n\n            train_preds  = self.PostProcessPreds(model.predict(Xtr))\n            tr_score     = self.ScoreMetric(ytr.values.flatten(), train_preds)\n            score        = self.ScoreMetric(ydev.values.flatten(), dev_preds)\n\n            scores.append(score)\n            tr_scores.append(tr_score)\n\n            nspace = 15 - len(method) - 2 if fold_nb <= 9 else 15 - len(method) - 1\n            \n            if self.es_req:\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} | Iter = {best_iter:,.0f} \")\n            else:\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f}\")\n                \n            mdl_best_iter.append(best_iter)\n\n            if test_preds_req:\n                test_preds.append(\n                    self.PostProcessPreds(\n                        model.predict(Xtest)\n                    )\n                )\n            else:\n                pass\n\n        test_preds    = np.mean(np.stack(test_preds, axis = 1), axis=1)\n        ftreimp       = pd.Series(ftreimp, index = Xdev.columns)\n        mdl_best_iter = np.uint16(np.amax(mdl_best_iter))\n\n        if ftreimp_plot_req :\n            print()\n            self.PlotFtreImp(ftreimp, method = method, ntop = ntop,)\n        else:\n            pass\n\n        PrintColor(f\"\\n---> {np.mean(scores):.6f} +- {np.std(scores):.6f} | OOF\", color = Fore.RED)\n        PrintColor(f\"---> {np.mean(tr_scores):.6f} +- {np.std(tr_scores):.6f} | Train\", color = Fore.RED)\n\n        if self.es_req == False:\n            pass\n        else:\n            PrintColor(f\"---> Max best iteration = {mdl_best_iter :,.0f}\", color = Fore.RED)\n\n        return (fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter)\n\n    def MakeOnlineModel(\n        self, X, y, Xtest, model, method,\n        test_preds_req : bool = False,\n    ):\n        \"This method refits the model on the complete train data and returns the model fitted object and predictions\"\n\n        try:\n            model.early_stopping_rounds = None\n        except:\n            pass\n\n        try:\n            model.fit(X, y, verbose = 0)\n        except:\n            model.fit(X, y,)\n\n        oof_preds  = model.predict(X)\n        if test_preds_req:\n            test_preds = model.predict(Xtest[X.columns])\n        else:\n            test_preds = 0\n        return (model, oof_preds, test_preds)\n\n    def MakeOfflinePreds(self, X, fitted_models):\n        \"This method creates test-set predictions for the offline model provided\"\n\n        test_preds = 0\n        n_splits   = len(fitted_models)\n        PrintColor(f\"---> Number of splits = {n_splits}\")\n\n        for model in fitted_models:\n            test_preds = test_preds + (model.predict(X) / n_splits)\n\n        return test_preds","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass OptunaEnsembler:\n    \"\"\"\n    This is the Optuna ensemble class-\n    Source- https://www.kaggle.com/code/arunklenin/ps3e26-cirrhosis-survial-prediction-multiclass\n    \"\"\";\n\n    def __init__(\n        self, state: int = 42, ntrials: int = 300, \n        metric_obj: str  = \"maximize\", \n        metric_lbl: str  = \"kappa\",\n        **params\n    ):\n        self.study        = None\n        self.weights      = None\n        self.random_state = state\n        self.n_trials     = ntrials\n        self.direction    = metric_obj\n        self.metric_lbl   = metric_lbl\n\n    def ScoreMetric(self, ytrue, ypred)->float:\n        \"\"\"\n        This is the metric function for the competition\n        \"\"\"\n        \n        if self.metric_lbl == \"rmse\":\n            return mse(ytrue, ypred, squared = False)\n        else:\n            myscore = \\\n            cohen_kappa_score(\n                np.uint8(np.around(ytrue,0)),\n                np.uint8(np.around(ypred,0)), \n                weights = \"quadratic\"\n            )\n            return myscore\n\n    def _objective(\n        self, trial, y_true, y_preds\n    ):\n        \"\"\"\n        This method defines the objective function for the ensemble\n        \"\"\";\n\n        if isinstance(y_preds, pd.DataFrame) or isinstance(y_preds, np.ndarray):\n            weights = [trial.suggest_float(f\"weight{n}\", 0.001, 0.999)\n                       for n in range(y_preds.shape[-1])\n                      ]\n            axis = 1\n\n        elif isinstance(y_preds, list):\n            weights = [trial.suggest_float(f\"weight{n}\", 0.001, 0.999)\n                       for n in range(len(y_preds))\n                      ]\n            axis = 0\n\n        # Calculating the weighted prediction:-\n        weighted_pred  = np.average(np.array(y_preds), axis = axis, weights = weights)\n        score          = self.ScoreMetric(y_true, weighted_pred)\n        return score\n\n    def fit(self, y_true, y_preds):\n        \"This method fits the Optuna objective on the fold level data\";\n\n        optuna.logging.set_verbosity = optuna.logging.ERROR\n\n        self.study = \\\n        optuna.create_study(sampler    = TPESampler(seed = self.random_state),\n                            pruner     = HyperbandPruner(),\n                            study_name = \"Ensemble\",\n                            direction  = self.direction,\n                           )\n\n        obj = partial(self._objective, y_true = y_true, y_preds = y_preds)\n        self.study.optimize(obj, n_trials = self.n_trials)\n\n        if isinstance(y_preds, list):\n            self.weights = [self.study.best_params[f\"weight{n}\"] for n in range(len(y_preds))]\n\n        else:\n            self.weights = [self.study.best_params[f\"weight{n}\"] for n in range(y_preds.shape[-1])]\n\n    def predict(self, y_preds):\n        \"This method predicts using the fitted Optuna objective\";\n\n        assert self.weights is not None, 'OptunaWeights error, must be fitted before predict';\n\n        if isinstance(y_preds, list):\n            weighted_pred = np.average(np.array(y_preds), axis=0, weights = self.weights)\n\n        else:\n            weighted_pred = np.average(np.array(y_preds), axis=1, weights = self.weights)\n\n        return weighted_pred\n\n    def fit_predict(self, y_true, y_preds):\n        \"\"\"\n        This method fits the Optuna objective on the fold data, then predicts the test set\n        \"\"\";\n        self.fit(y_true, y_preds)\n        return self.predict(y_preds)\n\n    def weights(self):\n        \"This method returns the non-normalized weights for all models in a fold\"\n        return self.weights\n\nprint()\ncollect();","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\ndef NormWeights(weights: dict, methods: list):\n    \"This function normalizes the weights and returns a dataframe of normalized weights across folds and models\"\n\n    weights = pd.DataFrame.from_dict(weights).T\n    weights[\"row_sum\"] = weights.sum(axis=1)\n\n    for col in weights.columns:\n        weights[col] = weights[col] / weights[\"row_sum\"]\n\n    weights.drop(\"row_sum\", axis = 1, inplace = True, errors = \"ignore\")\n    weights.columns    = methods\n    weights.index.name = \"Fold_Nb\"\n    return weights\n\ndef MakeEnsemble(\n    target: str, metric_obj: str, metric_lbl: str = \"kappa\"\n):\n    \"This function implements the Optuna ensemble on the OOF and test prediction datasets\"\n\n    global OOF_Preds, Mdl_Preds\n\n    PrintColor(f\"\\n{'=' * 20} ENSEMBLE {'=' * 20}\\n\")\n    \n    ygrp       = OOF_Preds[\"fold_nb\"]\n    cv         = PDS(ygrp)\n    oof_preds  = np.zeros(len(OOF_Preds))\n    test_preds = []\n    scores     = []\n    weights    = {}\n    drop_cols  = [\"fold_nb\", target, \"Ensemble\"]\n    n_splits   = ygrp.nunique()\n\n    for fold_nb, (_, dev_idx) in tqdm(enumerate(cv.split(OOF_Preds, OOF_Preds[target]))):\n        Xdev = OOF_Preds.iloc[dev_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        ydev = OOF_Preds.loc[dev_idx, target]\n\n        ens = OptunaEnsembler(\n            ntrials = CFG.ntrials, metric_lbl = metric_lbl, metric_obj = metric_obj\n        )\n        ens.fit(ydev, Xdev,)\n\n        dev_preds = ens.predict(Xdev)\n        score     = ens.ScoreMetric(ydev.values, dev_preds)\n        oof_preds[dev_idx] = dev_preds\n        test_preds.append(\n            ens.predict(Mdl_Preds.drop(drop_cols, axis=1, errors = \"ignore\"))\n        )\n\n        PrintColor(f\"---> {score: .6f} | Fold {fold_nb}\", color = Fore.CYAN)\n        scores.append(score)\n\n        weights[f\"Fold{fold_nb}\"] = ens.weights\n\n    PrintColor(f\"\\n---> OOF = {np.mean(scores): .6f} +- {np.std(scores): .6f} | Ensemble\",\n               color = Fore.RED\n              )\n\n    test_preds = np.mean(np.stack(test_preds, axis=1), axis=1,)\n\n    OOF_Preds[\"Ensemble\"] = oof_preds\n    Mdl_Preds[\"Ensemble\"] = test_preds\n\n    weights = \\\n    NormWeights(\n        weights,\n        methods = Mdl_Preds.drop(drop_cols, axis=1, errors = \"ignore\").columns\n    )\n\n    print(\"\\n\\n\\n\")\n    display(\n        weights.\\\n        style.\\\n        set_caption(\"Normalized weights\").\\\n        format(precision = 6).\\\n        set_properties(\n            props = \"color:red; background-color:white; font-weight: bold; border: maroon dashed 1.6px\"\n        )\n    )\n    \n    print()\n    display(\n        weights.mean().to_frame().transpose().\\\n        style.\\\n        format(precision = 6).\\\n        set_caption(\"Normalized Mean weights\").\\\n        set_properties(\n            props = \"color:red; background-color:white; font-weight: bold; border: maroon dashed 1.6px\"\n        )        \n    )\n\n    return weights\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass OptimizedRounder:\n    \"\"\"\n    Source - https://www.kaggle.com/code/tubotubo/starter-notebook-multi-target-prediction\n    \"\"\"\n\n    def __init__(\n        self, n_classes: int, n_trials: int = 100, direction : str = \"maximize\"\n    ):\n        self.n_classes  = n_classes\n        self.labels     = np.arange(n_classes)\n        self.n_trials   = n_trials\n        self.metric     = partial(cohen_kappa_score, weights=\"quadratic\")\n        self.direction  = direction\n        \n    def _objective(\n        self, trial: optuna.Trial, y_true: NDArray[np.int_], y_pred: NDArray[np.float_],\n    ) -> float:\n        \n        thresholds = []\n        for i in range(self.n_classes - 1):\n            low  = max(thresholds) if i > 0 else min(self.labels)\n            high = max(self.labels)\n            th   = trial.suggest_float(f\"threshold_{i}\", low, high)\n            thresholds.append(th)\n            \n        try:\n            y_pred_rounded = np.digitize(y_pred, thresholds)\n        except ValueError:\n            return -100\n        return self.metric(y_true, y_pred_rounded)\n\n    def fit(\n        self, y_pred: NDArray[np.float_], y_true: NDArray[np.int_]\n    ) -> None:\n        y_pred = self._normalize(y_pred)\n        study  = optuna.create_study(direction = self.direction)\n        obj    = partial(self._objective, y_true = y_true, y_pred = y_pred)\n        \n        study.optimize(obj, n_trials = self.n_trials)\n        self.thresholds = [study.best_params[f\"threshold_{i}\"] for i in range(self.n_classes - 1)]\n\n    def predict(self, y_pred: NDArray[np.float_]) -> NDArray[np.int_]:\n        assert hasattr(self, \"thresholds\"), \"fit() must be called before predict()\"\n        y_pred = self._normalize(y_pred)\n        return np.digitize(y_pred, self.thresholds)\n\n    def _normalize(self, y: NDArray[np.float_]) -> NDArray[np.float_]:\n        return (y - y.min()) / (y.max() - y.min()) * (self.n_classes - 1)\n    \n    def thresholds(self):\n        return self.thresholds\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile -a training.py\n\na = 2.998\nb = 1.092\n\nif LAUNCH_VARIANT == 'option 37': \n    a = 2.998\n    b = 1.095\nif LAUNCH_VARIANT == 'option 38':\n    a = 2.998\n    b = 1.089\n    \nif LAUNCH_VARIANT == 'option 39':\n    a = 3.001\n    b = 1.089\nif LAUNCH_VARIANT == 'option 40':\n    a = 2.995\n    b = 1.089\n    \nif LAUNCH_VARIANT == 'option 41':\n    a,b = 2.999,1.089\nif LAUNCH_VARIANT == 'option 42':\n    a,b = 2.997,1.089   \nif LAUNCH_VARIANT == 'option 43':\n    a,b = 2.998,1.088\n    \nif LAUNCH_VARIANT == 'option 45':\n    a,b = 2.999,1.089\nif LAUNCH_VARIANT == 'option 46':\n    a,b = 2.998,1.088   \nif LAUNCH_VARIANT == 'option 47':\n    a,b = 2.997,1.087\n    \nif LAUNCH_VARIANT == 'option 48':\n    a,b = 2.998,1.088\nif LAUNCH_VARIANT == 'option 49':\n    a,b = 2.998,1.088    \n    \nif LAUNCH_VARIANT == 'option 50':\n    a,b = 2.998,1.088\nif LAUNCH_VARIANT == 'option 51':\n    a,b = 2.998,1.088    \nif LAUNCH_VARIANT == 'option 52':\n    a,b = 2.998,1.088\nif LAUNCH_VARIANT == 'option 53':\n    a,b = 2.998,1.088  \nif LAUNCH_VARIANT == 'option 54':\n    a,b = 2.998,1.088    \n\nif LAUNCH_VARIANT == 'option 59':\n    a,b = 2.998,1.088\n    \nif LAUNCH_VARIANT == 'option 60':\n    a,b = 2.998,1.0883\nif LAUNCH_VARIANT == 'option 61':\n    a,b = 2.998,1.0884\nif LAUNCH_VARIANT == 'option 62':\n    a,b = 2.998,1.0885\n    \ndef MakeObj(y_true, y_pred):\n    \"This function is the common custom objective for LGBM and XGB\"\n    \n    labels = y_true + a\n    preds  = y_pred + a\n    preds  = preds.clip(0, np.inf)\n    f      = 1/2*np.sum((preds-labels)**2)\n    g      = 1/2*np.sum((preds-a)**2+b)\n    df     = preds - labels\n    dg     = preds - a\n    grad   = (df/g - f*dg/g**2)*len(labels)\n    hess   = np.ones(len(labels))\n    \n    return grad, hess\n\ndef MakeEvalMetric(y_true, y_pred):\n    \n    if isinstance(y_pred, xgb.QuantileDMatrix):\n        y_true, y_pred = y_pred, y_true\n        y_true = (y_true.get_label() + a).round()\n        y_pred = (y_pred + a).clip(0, np.inf).round()\n        \n        qwk = cohen_kappa_score(np.int8(y_true), np.int8(y_pred), weights=\"quadratic\")\n        return 'QWK', qwk\n\n    else:\n        y_true = y_true + a\n        y_pred = (y_pred + a).clip(0, np.inf).round()\n        \n        qwk = cohen_kappa_score(np.int8(y_true), np.int8(y_pred), weights=\"quadratic\")\n        return 'QWK', qwk, True","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n\nexec(open('training.py','r').read())\n\ntry:\n    l = MyLogger()\n    l.init(logging_lbl = \"lightgbm_custom\")\n    lgb.register_logger(l)\nexcept:\n    pass\n\n# Initializing CV scheme\ncv = cv_selector[CFG.mdlcv_mthd]\n\n# Initializing model parameters\nMdl_Master = \\\n{\n f'LGBM1R' : LGBMR(**{\"objective\"           : MakeObj,\n                      \"metrics\"             : \"None\",\n                      'device'              : \"gpu\" if CFG.gpu_switch == \"ON\" else \"cpu\",\n                      'learning_rate'       : 0.025, \n                      'n_estimators'        : 270,\n                      'max_depth'           : 6, \n                      'num_leaves'          : 85, \n                      'min_data_in_leaf'    : 12,\n                      'feature_fraction'    : 0.70, \n                      'bagging_fraction'    : 0.88, \n                      'bagging_freq'        : 6, \n                      'lambda_l1'           : 9.92, \n                      'lambda_l2'           : 4.35,\n                      'verbosity'           : -1,\n                      'random_state'        : CFG.state,\n                     }\n                  ),\n    \n f'XGB1R' : XGBR(**{  \"objective\"           : MakeObj,\n                      'device'              : \"cuda\" if CFG.gpu_switch == \"ON\" else \"cpu\",\n                      'learning_rate'       : 0.03, \n                      'n_estimators'        : 200,\n                      'max_depth'           : 4, \n                      'colsample_bytree'    : 0.55, \n                      'colsample_bynode'    : 0.60,\n                      'colsample_bylevel'   : 0.70,                     \n                      'reg_alpha'           : 2.50, \n                      'reg_lambda'          : 7.50,\n                      'verbose'             : 0,\n                      'random_state'        : CFG.state,\n                      'enable_categorical'  : True,\n                      'callbacks'           : [XGBLogging(epoch_log_interval= 0)],\n                     }\n                  ),\n}\n\n# Initializing model outputs\nOOF_Preds    = {}\nMdl_Preds    = {}\nFittedModels = {}\nFtreImp      = {}\nSelMdlCols   = {}","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nmytrain  = train.to_pandas().dropna(subset = [CFG.target])\nmytest   = test.to_pandas()[sel_cols]\nmytarget = CFG.target\n\nmytrain.index = range(len(mytrain))\nPrintColor(f\"---> Shapes = {mytrain.shape} {mytest.shape}\", color = Fore.CYAN)\n\n# Initializing CV folds across the training data:-\nfolds = np.zeros(len(mytrain))\nfor fold_nb, (train_idx, dev_idx) in enumerate(cv.split(mytrain, mytrain[mytarget])):\n    folds[dev_idx] = fold_nb\nmytrain[\"fold_nb\"] = folds\ndel folds\n\nPrintColor(f\"---> Shapes = {mytrain.shape} {mytest.shape}\\n\\n\", color = Fore.CYAN)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n      \n# Creating CV scheme:-\nmd = ModelTrainer(es = CFG.nbrnd_erly_stp, target = mytarget)\n\nfor method, mdl in tqdm(Mdl_Master.items()):\n    PrintColor(f\"\\n{'-' * 10} {method} MODEL TRAINING - {mytarget} {'-' * 10}\\n\", \n               color = Fore.MAGENTA\n              )\n\n    fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter =  \\\n    md.MakeOfflineModel(\n        mytrain[sel_cols],\n        mytrain[mytarget],\n        mytrain[\"fold_nb\"],\n        mytest,\n        clone(mdl),\n        method,\n        test_preds_req   = True,\n        ftreimp_plot_req = True,\n        ntop = 50,\n    ) \n\n    # Integrating data    \n    OOF_Preds[f\"{method}\"]    = oof_preds\n    Mdl_Preds[f\"{method}\"]    = test_preds\n    FtreImp[f\"{method}\"]      = ftreimp\n    FittedModels[f\"{method}\"] = fitted_models\n\n    del fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter\n    _ = utils.CleanMemory()\n    \nPrintColor(utils.CleanMemory())    ","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nOOF_Preds = pd.DataFrame.from_dict(OOF_Preds, orient = \"columns\")\nMdl_Preds = pd.DataFrame.from_dict(Mdl_Preds, orient = \"columns\")\n\nOOF_Preds[mytarget]  = mytrain[mytarget].values\nOOF_Preds[\"fold_nb\"] = mytrain[\"fold_nb\"].values\n\nweights = \\\nMakeEnsemble(target = mytarget, metric_obj = CFG.metric_obj, metric_lbl = 'kappa')\n\nprint()\nPrintColor(utils.CleanMemory())","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nmytuner = OptimizedRounder(n_classes = 4, n_trials = CFG.ntrials)\nytrain  = np.uint8(mytrain[CFG.target])\n\nmytuner.fit(OOF_Preds[\"LGBM1R\"], ytrain)\nens_preds  = mytuner.predict(OOF_Preds[\"LGBM1R\"])\ntest_preds = mytuner.predict(Mdl_Preds[\"LGBM1R\"])\n\nPrintColor(f\"---> Thresholds for labels\")\nwith np.printoptions(linewidth = 100, precision = 5):\n    pprint(np.array(mytuner.thresholds))\n\n# Displaying the confusion matrix \nscore = utils.ScoreMetric(ytrain, ens_preds)\nPrintColor(f\"\\n---> Final ensemble OOF score = {score :.6f}\\n\\n\")\n\nfig, ax = plt.subplots(1,1, figsize = (5,5))\ndisp = \\\nConfusionMatrixDisplay(\n    confusion_matrix = confusion_matrix(ytrain, ens_preds),  \n    display_labels = list(range(4))\n)\n\ndisp.plot(\n    cmap = \"Blues\", \n    ax = ax, \n    colorbar = False, \n    xticks_rotation = 0,\n    text_kw = {\"fontweight\": \"bold\",  \n               \"fontsize\"  : 16,\n              }\n)\nax.set_title(\n    f\"Confusion matrix - CV = {score :.6f}\", **CFG.title_specs\n)\nax.grid(**CFG.grid_specs)\nax.set(ylabel = f\"True {CFG.target}\", \n       xlabel = f\"Predicted {CFG.target}\", \n      )\nplt.show()\n\n_ = utils.CleanMemory()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\ntry:\n    print()\n    display(\n        OOF_Preds.head().style.format(precision = 3).set_caption(\"OOF Predictions\")\n    )\n\n    print()\n    display(\n        Mdl_Preds.head().style.format(precision = 3).set_caption(\"Model Predictions\")\n    )\nexcept:\n    pass\n\nsub_fl.with_columns(\n    pl.Series(CFG.target, test_preds.flatten(), pl.UInt8)\n).write_csv(\"submission_11.csv\")\n\nprint()\n!ls \nprint(f\"\\n\\n---> Submission file\\n\\n\")\n!head submission.csv\n\nPrintColor(utils.CleanMemory())","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_13' in ENSEMBLE_SOLUTIONS:\n    \n    import numpy as np\n    import polars as pl\n    import pandas as pd\n    from sklearn.base import clone\n    from copy import deepcopy\n    import optuna\n    from scipy.optimize import minimize\n\n    import re\n    from colorama import Fore, Style\n\n    from tqdm import tqdm\n    from IPython.display import clear_output\n\n    import warnings\n    warnings.filterwarnings('ignore')\n    pd.options.display.max_columns = None\n\n    import lightgbm as lgb\n    from catboost import CatBoostRegressor, CatBoostClassifier\n    from xgboost import XGBRegressor\n    from sklearn.ensemble import VotingRegressor\n    from sklearn.model_selection import *\n    from sklearn.metrics import *\n    import os\n\n    SEED = 42\n    n_splits = 5","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_13' in ENSEMBLE_SOLUTIONS:\n\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n    def load_time_series(dirname) -> pd.DataFrame:\n        ids = os.listdir(dirname)\n        indexes = []\n        stats = []\n        for idname in tqdm(ids):\n            df = pd.read_parquet(os.path.join(dirname, idname, 'part-0.parquet'))\n            df.drop('step', axis=1, inplace=True)\n            stats.append(df.describe().iloc[1:].values.reshape(-1))\n            indexes.append(idname.split('=')[1])\n        df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n        df['id'] = indexes\n        return df\n\n    train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n    test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n    time_series_cols = train_ts.columns.tolist()\n    time_series_cols.remove(\"id\")\n\n    train = pd.merge(train, train_ts, how=\"left\", on='id')\n    test = pd.merge(test, test_ts, how=\"left\", on='id')\n\n    train = train.drop('id',axis=1)\n    test = test.drop('id',axis=1)\n\n    featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\n\n    featuresCols += time_series_cols\n\n    train = train[featuresCols]\n    train = train.dropna(subset='sii')\n\n    cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\n    def update(df):\n        global cat_c\n        for c in cat_c : \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n\n        return df\n\n    train = update(train)\n    test = update(test)\n\n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n\n    for col in cat_c:\n\n        mapping = create_mapping(col, train)\n        mappingTe = create_mapping(col, test)\n\n        train[col] = train[col].replace(mapping).astype(int)\n        test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_13' in ENSEMBLE_SOLUTIONS:\n\n    def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n    def threshold_Rounder(oof_non_rounded, thresholds):\n        return np.where(oof_non_rounded < thresholds[0], 0,\n                        np.where(oof_non_rounded < thresholds[1], 1,\n                                 np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n    def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n        rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_p)\n\n    def TrainML(model_class, test_data, tune=False):\n\n        X = train.drop(['sii'], axis=1)\n        y = train['sii']\n\n        SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n        train_S = []\n        test_S = []\n\n        oof_non_rounded = np.zeros(len(y), dtype=float) \n        oof_rounded = np.zeros(len(y), dtype=int) \n        test_preds = np.zeros((len(test_data), n_splits))\n\n        for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n            model = clone(model_class)\n            model.fit(X_train, y_train)\n\n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n\n            oof_non_rounded[test_idx] = y_val_pred\n            y_val_pred_rounded = y_val_pred.round(0).astype(int)\n            oof_rounded[test_idx] = y_val_pred_rounded\n\n            train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n            val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n            train_S.append(train_kappa)\n            test_S.append(val_kappa)\n\n            test_preds[:, fold] = model.predict(test_data)\n\n            print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            clear_output(wait=True)\n\n        print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n        print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n        KappaOPtimizer = minimize(evaluate_predictions,\n                                  x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                                  method='Nelder-Mead') # Nelder-Mead | # Powell\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n        tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n        print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n        if tune:\n            return np.mean(test_S)\n\n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n        submission = pd.DataFrame({\n            'id': sample['id'],\n            'sii': tpTuned\n        })\n        return submission","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_13' in ENSEMBLE_SOLUTIONS: pass\n\n    # # 自定义目标函数 Custom objective functions\n    # def objective(trial):\n    #     # 定义需要调节的超参数空间 Define the hyperparameter space to be adjusted\n    #     params = {\n    #         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-4, 1e-1),\n    #         'max_depth': trial.suggest_int('max_depth', 3, 12),\n    #         'num_leaves': trial.suggest_int('num_leaves', 31, 512),\n    #         'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 10, 100),\n    #         'feature_fraction': trial.suggest_uniform('feature_fraction', 0.5, 1.0),\n    #         'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.5, 1.0),\n    #         'bagging_freq': trial.suggest_int('bagging_freq', 1, 7),\n    #         'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-8, 10.0),\n    #         'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-8, 10.0),\n    #     }\n\n    #     Light = lgb.LGBMRegressor(**params,random_state=SEED, verbose=-1,n_estimators=200)\n    #     return TrainML(Light, test, True)\n\n    # # 创建一个Optuna Study Create an Optuna Study\n    # study = optuna.create_study(direction='maximize')  \n    # study.optimize(objective, n_trials=100)  # 设置进行100次搜索 Set 100 searches\n\n    # # 输出最优超参数和最优结果 Output optimal hyperparameters and optimal results\n    # print(\"Best trial:\")\n    # trial = study.best_trial\n\n    # print(f\"   {trial.value}\")\n    # print(\"  Best hyperparameters: \", trial.params)\n\n    # # Best trial:\n    # #    0.48050601911401003\n    # #   Best hyperparameters:  {'learning_rate': 0.01807927986490293, 'max_depth': 5, 'num_leaves': 309, 'min_data_in_leaf': 87, 'feature_fraction': 0.5258796792372337, 'bagging_fraction': 0.8961353601288653, 'bagging_freq': 3, 'lambda_l1': 0.0012794381687055592, 'lambda_l2': 4.7645510836117515e-06}\n\n    # # Best trial:\n    # #    0.40180590915921055\n    # #   Best hyperparameters:  {'learning_rate': 0.09056475084257094, 'max_depth': 5, 'num_leaves': 429, 'min_data_in_leaf': 43, 'feature_fraction': 0.8916815865803562, 'bagging_fraction': 0.8276271977726875, 'bagging_freq': 6, 'lambda_l1': 0.07688998999378223, 'lambda_l2': 0.00026663802273190103}\n","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_13' in ENSEMBLE_SOLUTIONS:\n\n    # Params = {'learning_rate': 0.07975474666326936, 'max_depth': 10, 'num_leaves': 207, 'min_data_in_leaf': 41,\n    #                 'feature_fraction': 0.6385678848225935, 'bagging_fraction': 0.9042038292349021, 'bagging_freq': 6, \n    #                             'lambda_l1': 9.920617415343463, 'lambda_l2': 4.351491475117983} # LB : 0.452\n    Params = {'learning_rate': 0.09056475084257094, 'max_depth': 5, 'num_leaves': 429, 'min_data_in_leaf': 43, \n              'feature_fraction': 0.8916815865803562, 'bagging_fraction': 0.8276271977726875, 'bagging_freq': 6, \n              'lambda_l1': 0.07688998999378223, 'lambda_l2': 0.00026663802273190103}\n\n\n    Light = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1,n_estimators=200)\n    Submission = TrainML(Light,test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_13' in ENSEMBLE_SOLUTIONS:\n\n    Submission.to_csv('submission_13.csv', index=False)\n    Submission.head()\n    print(Submission['sii'].value_counts())","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_14' in ENSEMBLE_SOLUTIONS: \n    \n    import numpy as np\n    import polars as pl\n    import pandas as pd\n    from sklearn.base import clone\n    from copy import deepcopy\n    import optuna\n    from scipy.optimize import minimize\n    import os\n\n    import re\n    from colorama import Fore, Style\n\n    from tqdm import tqdm\n    from IPython.display import clear_output\n    from concurrent.futures import ThreadPoolExecutor\n\n    import warnings\n    warnings.filterwarnings('ignore')\n    pd.options.display.max_columns = None\n\n    import lightgbm as lgb\n    from catboost import CatBoostRegressor, CatBoostClassifier\n    from xgboost import XGBRegressor\n    from sklearn.ensemble import VotingRegressor\n    from sklearn.model_selection import *\n    from sklearn.metrics import *\n\n    SEED = 42\n    n_splits = 5","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:\n\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n    def process_file(filename, dirname):\n        df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n        df.drop('step', axis=1, inplace=True)\n        return df.describe().values.reshape(-1), filename.split('=')[1]\n\n    def load_time_series(dirname) -> pd.DataFrame:\n        ids = os.listdir(dirname)\n\n        with ThreadPoolExecutor() as executor:\n            results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n        stats, indexes = zip(*results)\n\n        df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n        df['id'] = indexes\n\n        return df\n\n    train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n    test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n    time_series_cols = train_ts.columns.tolist()\n    time_series_cols.remove(\"id\")\n\n    train = pd.merge(train, train_ts, how=\"left\", on='id')\n    test = pd.merge(test, test_ts, how=\"left\", on='id')\n\n    train = train.drop('id',axis=1)\n    test = test.drop('id',axis=1)\n\n    featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\n\n    featuresCols += time_series_cols\n\n    train = train[featuresCols]\n    train = train.dropna(subset='sii')\n\n    cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\n    def update(df):\n\n        global cat_c\n        for c in cat_c : \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n\n        return df\n\n    train = update(train)\n    test = update(test)\n\n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n\n    for col in cat_c:\n\n        mapping = create_mapping(col, train)\n        mappingTe = create_mapping(col, test)\n\n        train[col] = train[col].replace(mapping).astype(int)\n        test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:\n\n    def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n    def threshold_Rounder(oof_non_rounded, thresholds):\n        return np.where(oof_non_rounded < thresholds[0], 0,\n                        np.where(oof_non_rounded < thresholds[1], 1,\n                                 np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n    def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n        rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_p)\n\n    def TrainML(model_class, test_data):\n\n        X = train.drop(['sii'], axis=1)\n        y = train['sii']\n\n        SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n        train_S = []\n        test_S = []\n\n        oof_non_rounded = np.zeros(len(y), dtype=float) \n        oof_rounded = np.zeros(len(y), dtype=int) \n        test_preds = np.zeros((len(test_data), n_splits))\n\n        for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n            model = clone(model_class)\n            model.fit(X_train, y_train)\n\n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n\n            oof_non_rounded[test_idx] = y_val_pred\n            y_val_pred_rounded = y_val_pred.round(0).astype(int)\n            oof_rounded[test_idx] = y_val_pred_rounded\n\n            train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n            val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n            train_S.append(train_kappa)\n            test_S.append(val_kappa)\n\n            test_preds[:, fold] = model.predict(test_data)\n\n            print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            clear_output(wait=True)\n\n        print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n        print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n        KappaOPtimizer = minimize(evaluate_predictions,\n                                  x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                                  method='Nelder-Mead') # Nelder-Mead | # Powell\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n        tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n        print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n        submission = pd.DataFrame({\n            'id': sample['id'],\n            'sii': tpTuned\n        })\n\n        return submission","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:\n\n    Params = {'learning_rate': 0.04603534510792164, 'max_depth': 12, 'num_leaves': 478, 'min_data_in_leaf': 13,\n              'feature_fraction': 0.8935304204489449, 'bagging_fraction': 0.7840117449237969, 'bagging_freq': 4,\n              'lambda_l1': 6.596560434072009, 'lambda_l2': 2.680080551210706e-06} \n\n    Light = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1,n_estimators=200)\n    Submission = TrainML(Light,test)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:\n\n    Submission.to_csv('submission_14.csv', index=False)\n    print(Submission['sii'].value_counts())","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n    \n    import numpy as np\n    import polars as pl\n    import pandas as pd\n    from sklearn.base import clone\n    from copy import deepcopy\n    import optuna\n    from scipy.optimize import minimize\n    import os\n\n    import re\n    from colorama import Fore, Style\n\n    from tqdm import tqdm\n    from IPython.display import clear_output\n    from concurrent.futures import ThreadPoolExecutor\n\n    import warnings\n    warnings.filterwarnings('ignore')\n    pd.options.display.max_columns = None\n\n    import lightgbm as lgb\n    from catboost import CatBoostRegressor, CatBoostClassifier\n    from xgboost import XGBRegressor\n    from sklearn.ensemble import VotingRegressor\n    from sklearn.model_selection import *\n    from sklearn.metrics import *\n\n    SEED = 42\n    n_splits = 5","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n    def process_file(filename, dirname):\n        df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n        df.drop('step', axis=1, inplace=True)\n        return df.describe().values.reshape(-1), filename.split('=')[1]\n\n    def load_time_series(dirname) -> pd.DataFrame:\n        ids = os.listdir(dirname)\n\n        with ThreadPoolExecutor() as executor:\n            results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n        stats, indexes = zip(*results)\n\n        df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n        df['id'] = indexes\n\n        return df\n\n    train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n    test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n    time_series_cols = train_ts.columns.tolist()\n    time_series_cols.remove(\"id\")\n\n    train = pd.merge(train, train_ts, how=\"left\", on='id')\n    test = pd.merge(test, test_ts, how=\"left\", on='id')\n\n    train = train.drop('id',axis=1)\n    test = test.drop('id',axis=1)\n\n    featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\n\n    featuresCols += time_series_cols\n\n    train = train[featuresCols]\n    train = train.dropna(subset='sii')\n\n    cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\n    def update(df):\n\n        global cat_c\n        for c in cat_c : \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n\n        return df\n\n    train = update(train)\n    test = update(test)\n\n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n\n    for col in cat_c:\n\n        mapping = create_mapping(col, train)\n        mappingTe = create_mapping(col, test)\n\n        train[col] = train[col].replace(mapping).astype(int)\n        test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n    \n    # fe index check\n    for i in train.columns.to_list():\n        if i.startswith('stat_'):\n            print(f\"{train.columns.to_list().index(i)},{i}\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n    \n    \"\"\"\n    59,stat_0\n    154,stat_95\n    \"\"\"\n    for i in range(59,154,5):\n        train[f\"stat_mean{i}\"] = train.iloc[:, i:i+5].mean(axis=1)\n    #     train[f\"stat_sum{i}\"] = train.iloc[:, i:i+5].sum(axis=1)\n\n    train[f\"stat_mean_all\"] = train.iloc[:, 59:154].mean(axis=1)\n    # train[f\"stat_sum_all\"] = train.iloc[:, 59:154].sum(axis=1)\n\n    # ---------------------------------------------------------------- #\n    for i in range(58,153,5):\n        test[f\"stat_mean{i+1}\"] = test.iloc[:, i:i+5].mean(axis=1)\n    #     test[f\"stat_sum{i+1}\"] = test.iloc[:, i:i+5].sum(axis=1)\n\n    test[f\"stat_mean_all\"] = test.iloc[:, 58:153].mean(axis=1)\n    # test[f\"stat_sum_all\"] = test.iloc[:, 58:153].sum(axis=1)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n    \n    display(train.head(), test.head())","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n\n    def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n    def threshold_Rounder(oof_non_rounded, thresholds):\n        return np.where(oof_non_rounded < thresholds[0], 0,\n                        np.where(oof_non_rounded < thresholds[1], 1,\n                                 np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n    def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n        rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_p)\n\n    def TrainML(model_class, test_data):\n\n        X = train.drop(['sii'], axis=1)\n        y = train['sii']\n\n        SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n        models = []\n\n        train_S = []\n        test_S = []\n\n        oof_non_rounded = np.zeros(len(y), dtype=float) \n        oof_rounded = np.zeros(len(y), dtype=int) \n        test_preds = np.zeros((len(test_data), n_splits))\n\n        for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n            model = clone(model_class)\n            model.fit(X_train, y_train)\n\n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n\n            oof_non_rounded[test_idx] = y_val_pred\n            y_val_pred_rounded = y_val_pred.round(0).astype(int)\n            oof_rounded[test_idx] = y_val_pred_rounded\n\n            train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n            val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n            train_S.append(train_kappa)\n            test_S.append(val_kappa)\n\n            test_preds[:, fold] = model.predict(test_data)\n\n            print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            clear_output(wait=True)\n            models.append(model)\n\n        print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n        print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n        KappaOPtimizer = minimize(evaluate_predictions,\n                                  x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                                  method='Nelder-Mead') # Nelder-Mead | # Powell\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n        tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n        print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n        submission = pd.DataFrame({\n            'id': sample['id'],\n            'sii': tpTuned\n        })\n\n        return submission, models, X","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n\n    Params = {'learning_rate': 0.04603534510792164, 'max_depth': 12, 'num_leaves': 478, 'min_data_in_leaf': 13,\n              'feature_fraction': 0.8935304204489449, 'bagging_fraction': 0.7840117449237969, 'bagging_freq': 4,\n              'lambda_l1': 6.596560434072009, 'lambda_l2': 2.680080551210706e-06} \n\n    Light = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1,n_estimators=200)\n    Submission, models_train1, X = TrainML(Light,test)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n    \n    import matplotlib.pyplot as plt\n    importance_df = pd.DataFrame(np.sum([model.feature_importances_ for model in models_train1], axis=0), index=X.columns, columns=['importance']).sort_values('importance',ascending=True)\n    importance_df\n\n    fig = plt.figure(figsize=(10, 30))\n    plt.barh(importance_df.index, importance_df[\"importance\"], align=\"center\")\n    plt.title(\"Feature Importance\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n    \n    IMPOTTANCE_TH = 0\n    drop_fes = importance_df[importance_df[\"importance\"]<=IMPOTTANCE_TH].index.to_list()\n    drop_fes","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n    \n    train = train.drop(columns=drop_fes)\n    test = test.drop(columns=drop_fes)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n\n    Params = {'learning_rate': 0.04603534510792164, 'max_depth': 12, 'num_leaves': 478, 'min_data_in_leaf': 13,\n              'feature_fraction': 0.8935304204489449, 'bagging_fraction': 0.7840117449237969, 'bagging_freq': 4,\n              'lambda_l1': 6.596560434072009, 'lambda_l2': 2.680080551210706e-06} \n\n    Light = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1,n_estimators=200)\n    Submission, models_train1, X = TrainML(Light,test)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n    \n    import matplotlib.pyplot as plt\n    importance_df = pd.DataFrame(np.sum([model.feature_importances_ for model in models_train1], axis=0), index=X.columns, columns=['importance']).sort_values('importance',ascending=True)\n    importance_df\n\n    fig = plt.figure(figsize=(10, 30))\n    plt.barh(importance_df.index, importance_df[\"importance\"], align=\"center\")\n    plt.title(\"Feature Importance\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_15' in ENSEMBLE_SOLUTIONS:\n\n    Submission.to_csv('submission_15.csv', index=False)\n    print(Submission['sii'].value_counts())","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:  \n    \n    import numpy as np\n    import polars as pl\n    import pandas as pd\n    from sklearn.base import clone\n    from copy import deepcopy\n    import optuna\n    from scipy.optimize import minimize\n    import matplotlib.pyplot as plt\n    import missingno as msno\n    import re\n    from colorama import Fore, Style\n\n    from tqdm import tqdm\n    from IPython.display import clear_output\n\n    import warnings\n    warnings.filterwarnings('ignore')\n    pd.options.display.max_columns = None\n\n    import lightgbm as lgb\n    from catboost import CatBoostRegressor, CatBoostClassifier\n    from xgboost import XGBRegressor\n    from catboost import CatBoostRegressor\n    import xgboost as xgb\n    from sklearn.ensemble import VotingRegressor\n    from sklearn.model_selection import *\n    from sklearn.metrics import *\n\n    SEED = 42\n    n_splits = 5","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n    train = train.drop('id',axis=1)\n    test = test.drop('id',axis=1)\n\n    featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\n\n    train = train[featuresCols]\n    #train = train.dropna(subset='sii')\n\n    cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\n    def update(df):\n        global cat_c\n        for c in cat_c : \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n\n        return df\n\n    #train = update(train)\n    #test = update(test)\n\n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n\n\n    for col in cat_c:\n        all_values = pd.concat([train[col], test[col]]).unique()\n        mapping = {value: idx for idx, value in enumerate(all_values)}\n\n        train[col] = train[col].replace(mapping).astype(int)\n        test[col] = test[col].replace(mapping).astype(int)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n\n    train","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS: \n    \n    import pandas as pd\n    import numpy as np\n    from lightgbm import LGBMRegressor, LGBMClassifier\n\n    def fill_missing_with_lgbm(train, test, target_column, n_estimators=500, random_state=42):\n        \"\"\"\n        使用LightGBM模型填补train和test的某个特征中的缺失值：\n        1. 先合并train和test，\n        2. 训练模型补全缺失值，\n        3. 再拆分出补全后的train和test。\n\n        参数:\n        train (pd.DataFrame): 训练集数据框\n        test (pd.DataFrame): 测试集数据框\n        target_column (str): 需要填补缺失值的目标特征（列名称）\n        model_type (str): 'regression' 或 'classification' 来指定模型类型\n        n_estimators (int): LightGBM基学习器数量\n        random_state (int): 随机种子\n\n        返回:\n        train_filled, test_filled: 两个DataFrame（train和test的缺失值已被填充）\n        \"\"\"\n        global cat_c\n        if target_column in cat_c:\n            model_type = 'classification'\n        else:\n            model_type = 'regression'\n\n        # 1. 添加标记列来标识 train 和 test\n        train['is_train'] = 1\n        test['is_train'] = 0\n\n        # 合并train和test数据集\n        df = pd.concat([train, test], ignore_index=True)\n\n        # 2. 找出不包含目标列的特征列\n        features_columns = df.columns[df.columns != target_column].tolist()\n\n        # 提取缺失值的行和不缺失的行\n        df_missing = df[df[target_column].isnull()]  # 缺失值部分\n        df_not_missing = df[~df[target_column].isnull()]  # 不缺失部分\n\n        if df_missing.empty or df_not_missing.empty:\n            print(f\"No missing data in '{target_column}' column or all data are missing.\")\n            return train, test\n\n        # 3. 准备训练集和特征\n        X_train = df_not_missing[features_columns]  # 非缺失值行的特征\n        y_train = df_not_missing[target_column]     # 非缺失值行的目标列\n\n        # 4. 初始化 LGBM 模型\n        if model_type == 'regression':\n            model = LGBMRegressor(n_estimators=n_estimators, random_state=random_state)\n        elif model_type == 'classification':\n            model = LGBMClassifier(n_estimators=n_estimators, random_state=random_state)\n        else:\n            raise ValueError(\"model_type should be either 'regression' or 'classification'.\")\n\n        # 5. 训练模型\n        model.fit(X_train, y_train)\n\n        # 6. 使用模型预测缺失值\n        X_pred = df_missing[features_columns]\n        y_pred = model.predict(X_pred)\n\n        # 7. 用预测值填充缺失值部分\n        df.loc[df[target_column].isnull(), target_column] = y_pred\n\n        # 8. 将数据集重新拆分为原来的 train 和 test\n        train_filled = df[df['is_train'] == 1].drop(columns=['is_train'])\n        test_filled = df[df['is_train'] == 0].drop(columns=['is_train'])\n\n        return train_filled, test_filled\n\n\n    tianbu_cols = ['CGAS-CGAS_Score','Physical-BMI','Physical-Height','Physical-Weight', \\\n                   'Physical-Diastolic_BP','Physical-HeartRate','Physical-Systolic_BP', \\\n                   'SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday']\n\n    for col in tianbu_cols:\n        print(\"开始填补特征：\"+ col)\n        train, test = fill_missing_with_lgbm(train, test, col)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    FGC_cols = [\n      'FGC-FGC_CU',\n      'FGC-FGC_CU_Zone',\n      'FGC-FGC_GSND',\n      'FGC-FGC_GSND_Zone',\n      'FGC-FGC_GSD',\n      'FGC-FGC_GSD_Zone',\n      'FGC-FGC_PU',\n      'FGC-FGC_PU_Zone',\n      'FGC-FGC_SRL',\n      'FGC-FGC_SRL_Zone',\n      'FGC-FGC_SRR',\n      'FGC-FGC_SRR_Zone',\n      'FGC-FGC_TL',\n      'FGC-FGC_TL_Zone'\n    ]\n    for col in FGC_cols:\n        print(\"开始填补特征：\"+ col)\n        train, test = fill_missing_with_lgbm(train, test, col)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    train = update(train)\n    test = update(test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    train.hist(figsize=(15, 10), bins=20, xlabelsize=8, ylabelsize=8)\n\n    plt.tight_layout()\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    train = train.dropna(subset='sii')\n    train[\"sii\"].hist()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    missing_percent_train = train.isnull().mean() * 100\n    missing_percent_train","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n\n    missing_percent_test = test.isnull().mean() * 100\n    missing_percent_test","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    msno.matrix(train)\n    plt.show()","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    msno.matrix(test)\n    plt.show()","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    def create_interaction_features(df, feature_pairs):\n        global cat_c\n        for feature1, feature2 in feature_pairs:\n            if(feature1 not in cat_c or feature2 not in cat_c):\n                print(\"feature1:\" + feature1 + \",feature2:\" + feature2)\n                new_feature_name = f\"{feature1}_x_{feature2}\"\n                df[new_feature_name] = df[feature1] * df[feature2]\n        return df","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    feature_pairs = [\n        ('PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age'),\n        ('Basic_Demos-Age', 'SDS-SDS_Total_T'),\n        ('FGC-FGC_SRR_Zone', 'SDS-SDS_Total_T'),\n        ('BIA-BIA_BMC', 'Physical-HeartRate'),\n        #('Fitness_Endurance-Season', 'Physical-Waist_Circumference'),\n        ('BIA-BIA_Fat', 'Physical-BMI'),\n        ('PreInt_EduHx-Season', 'Fitness_Endurance-Season'),\n        ('SDS-SDS_Total_T', 'Physical-Systolic_BP'),\n        ('Basic_Demos-Sex', 'FGC-FGC_PU_Zone')\n    ]\n\n    train = create_interaction_features(train, feature_pairs)\n    test = create_interaction_features(test, feature_pairs)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:    \n    \n    train[[\"Fitness_Endurance-Season\",\"Physical-Waist_Circumference\"]]","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    # モデルを学習した後のコード\n    XGBoost = xgb.XGBRegressor(random_state=SEED,enable_categorical=True)\n    XGBoost.fit(X, y)\n\n    # 特徴量重要度を取得\n    importance = XGBoost.feature_importances_\n\n    # 特徴量の名前を取得\n    features = X.columns\n\n    # データフレームとして整理\n    importance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\n\n    # 特徴量重要度を降順に並び替え\n    importance_df = importance_df.sort_values(by='Importance', ascending=False)\n\n    # プロット\n    plt.figure(figsize=(10, 20))\n    plt.barh(importance_df['Feature'], importance_df['Importance'])\n    plt.xlabel('Feature Importance')\n    plt.ylabel('Features')\n    plt.title('Feature Importance in XGBoost')\n    plt.gca().invert_yaxis()  # 重要度が高いものを上に\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    # 特徴量の名前を取得\n    features = X.columns\n\n    # データフレームとして整理（特徴量重要度と欠損値の割合を結合）\n    importance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\n    missing_df = pd.DataFrame({'Feature': missing_percent_train.index, 'MissingPercent': missing_percent_train.values})\n    combined_df = pd.merge(importance_df, missing_df, on='Feature')\n\n    # 散布図を作成\n    plt.figure(figsize=(10, 6))\n    plt.scatter(combined_df['MissingPercent'], combined_df['Importance'], alpha=0.7)\n    plt.xlabel('Missing Percentage (%)')\n    plt.ylabel('Feature Importance')\n    plt.title('Feature Importance vs Missing Percentage')\n    plt.grid(True)\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:    \n    \n    # 获取各自列的集合\n    train_columns = set(train.columns)\n    test_columns = set(test.columns)\n\n    # 查找只在 train 中但不在 test 中的列\n    train_only = train_columns - test_columns\n\n    # 查找只在 test 中但不在 train 中的列\n    test_only = test_columns - train_columns\n\n    # 输出差异列\n    print(f\"Columns only in train: {train_only}\")\n    print(f\"Columns only in test: {test_only}\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n\n    def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n    def threshold_Rounder(oof_non_rounded, thresholds):\n        return np.where(oof_non_rounded < thresholds[0], 0,\n                        np.where(oof_non_rounded < thresholds[1], 1,\n                                 np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n    def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n        rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_p)\n\n    def TrainML(model_class, test_data):\n\n        X = train.drop(['sii'], axis=1)\n        y = train['sii']\n        test_data = test_data.drop(['sii'], axis=1)\n        SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n        train_S = []\n        test_S = []\n\n        oof_non_rounded = np.zeros(len(y), dtype=float) \n        oof_rounded = np.zeros(len(y), dtype=int) \n        test_preds = np.zeros((len(test_data), n_splits))\n\n        for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n            model = clone(model_class)\n            model.fit(X_train, y_train)\n\n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n\n            oof_non_rounded[test_idx] = y_val_pred\n            y_val_pred_rounded = y_val_pred.round(0).astype(int)\n            oof_rounded[test_idx] = y_val_pred_rounded\n\n            train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n            val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n            train_S.append(train_kappa)\n            test_S.append(val_kappa)\n\n            test_preds[:, fold] = model.predict(test_data)\n\n            print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            clear_output(wait=True)\n\n        print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n        print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n        KappaOPtimizer = minimize(evaluate_predictions,\n                                  x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                                  method='Nelder-Mead') # Nelder-Mead | # Powell\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n        tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n        print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n        submission = pd.DataFrame({\n            'id': sample['id'],\n            'sii': tpTuned\n        })\n\n        return submission, tKappa","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    best_params_lgbm = {'learning_rate': 0.011747572224219955, 'n_estimators': 993,  'min_child_weight': 0.0025036281384857462, 'colsample_bytree': 0.6648896193058901, 'reg_alpha': 0.7153672744430527, 'reg_lambda': 0.12158717311465662}\n    best_params_xgb = {'learning_rate': 0.007356059931165658, 'max_depth': 3, 'n_estimators': 957, 'subsample': 0.6555266544650088, 'colsample_bytree': 0.7712019245727745}\n    best_params_catboost = {'iterations': 804, 'learning_rate': 0.007849710402582562, 'depth': 6, 'l2_leaf_reg': 7.31183636902306, 'subsample': 0.5630297785016092, 'random_strength': 1.7097065892440113, 'bagging_temperature': 0.026593521316435192, 'border_count': 12}","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    # XGBoost\n    XGBoost = xgb.XGBRegressor(**best_params_xgb, random_state=SEED,enable_categorical=True)\n    Submission_XGB, k_xgb = TrainML(XGBoost, test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    # CatBoost\n    CatBoost = CatBoostRegressor(**best_params_catboost, random_state=SEED, verbose=0,cat_features=cat_c)\n    Submission_CatBoost , k_cat= TrainML(CatBoost, test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    # LightGBM\n    Light = lgb.LGBMRegressor(**best_params_lgbm, random_state=SEED, verbose=-1)\n    Submission_LGBM, k_lgbm = TrainML(Light, test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS: \n    \n    print(Submission_LGBM['sii'].value_counts())\n    print(Submission_XGB['sii'].value_counts())\n    print(Submission_CatBoost['sii'].value_counts())","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    # k値の合計を計算して、各モデルの重みを計算\n    total_k = k_cat + k_xgb + k_lgbm\n\n    weight_cat = k_cat / total_k    # CatBoost の重み\n    weight_xgb = k_xgb / total_k    # XGBoost の重み\n    weight_lgbm = k_lgbm / total_k  # LightGBM の重み\n\n    # 各モデルの予測結果（submission）を用意\n    # 'sii' がカテゴリラベルであることを前提とします\n    ensemble_df = pd.DataFrame({\n        'id': Submission_LGBM['id'],\n        'cat': Submission_CatBoost[\"sii\"],\n        'xgb': Submission_XGB[\"sii\"],\n        'lgbm': Submission_LGBM[\"sii\"]\n    })\n\n    # 予測結果を長い形式に変換\n    melted = ensemble_df.melt(id_vars='id', value_vars=['cat', 'xgb', 'lgbm'], \n                              var_name='model', value_name='sii')\n\n    # 各モデルに対応する重みを割り当て\n    melted['weight'] = melted['model'].map({\n        'cat': weight_cat,\n        'xgb': weight_xgb,\n        'lgbm': weight_lgbm\n    })\n\n    # 各idごと、siiごとに重みを集計\n    grouped = melted.groupby(['id', 'sii'])['weight'].sum().reset_index()\n\n    # 各idごとに最大の重みを持つsiiを選択\n    best_submission = grouped.loc[grouped.groupby('id')['weight'].idxmax()][['id', 'sii']]","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    comparison_df = best_submission.merge(ensemble_df, on='id', how='left')","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n    \n    comparison_df","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_16' in ENSEMBLE_SOLUTIONS:\n\n    best_submission.to_csv('submission_16.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    import numpy as np\n    import polars as pl\n    import pandas as pd\n    from sklearn.base import clone\n    from copy import deepcopy\n    import optuna\n    from scipy.optimize import minimize\n    import matplotlib.pyplot as plt\n    import missingno as msno\n    import re\n    from colorama import Fore, Style\n\n    from tqdm import tqdm\n    from IPython.display import clear_output\n\n    import warnings\n    warnings.filterwarnings('ignore')\n    pd.options.display.max_columns = None\n\n    import lightgbm as lgb\n    from catboost import CatBoostRegressor, CatBoostClassifier\n    from xgboost import XGBRegressor\n    from catboost import CatBoostRegressor\n    import xgboost as xgb\n    from sklearn.ensemble import VotingRegressor\n    from sklearn.model_selection import *\n    from sklearn.metrics import *\n\n    SEED = 42\n    n_splits = 5","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n    train = train.drop('id',axis=1)\n    test = test.drop('id',axis=1)\n\n    featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\n\n    train = train[featuresCols]\n    #train = train.dropna(subset='sii')\n\n    cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\n    def update(df):\n        global cat_c\n        for c in cat_c : \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n\n        return df\n\n    #train = update(train)\n    #test = update(test)\n\n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n\n\n    for col in cat_c:\n        all_values = pd.concat([train[col], test[col]]).unique()\n        mapping = {value: idx for idx, value in enumerate(all_values)}\n\n        train[col] = train[col].replace(mapping).astype(int)\n        test[col] = test[col].replace(mapping).astype(int)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    train","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    import pandas as pd\n    import numpy as np\n    from lightgbm import LGBMRegressor, LGBMClassifier\n\n    def fill_missing_with_lgbm(train, test, target_column, n_estimators=1000, random_state=42):\n        \"\"\"\n        使用LightGBM模型填补train和test的某个特征中的缺失值：\n        1. 先合并train和test，\n        2. 训练模型补全缺失值，\n        3. 再拆分出补全后的train和test。\n\n        参数:\n        train (pd.DataFrame): 训练集数据框\n        test (pd.DataFrame): 测试集数据框\n        target_column (str): 需要填补缺失值的目标特征（列名称）\n        model_type (str): 'regression' 或 'classification' 来指定模型类型\n        n_estimators (int): LightGBM基学习器数量\n        random_state (int): 随机种子\n\n        返回:\n        train_filled, test_filled: 两个DataFrame（train和test的缺失值已被填充）\n        \"\"\"\n        global cat_c\n        if target_column in cat_c:\n            model_type = 'classification'\n        else:\n            model_type = 'regression'\n\n        # 1. 添加标记列来标识 train 和 test\n        train['is_train'] = 1\n        test['is_train'] = 0\n\n        # 合并train和test数据集\n        df = pd.concat([train, test], ignore_index=True)\n\n        # 2. 找出不包含目标列的特征列\n        features_columns = df.columns[df.columns != target_column].tolist()\n\n        # 提取缺失值的行和不缺失的行\n        df_missing = df[df[target_column].isnull()]  # 缺失值部分\n        df_not_missing = df[~df[target_column].isnull()]  # 不缺失部分\n\n        if df_missing.empty or df_not_missing.empty:\n            print(f\"No missing data in '{target_column}' column or all data are missing.\")\n            return train, test\n\n        # 3. 准备训练集和特征\n        X_train = df_not_missing[features_columns]  # 非缺失值行的特征\n        y_train = df_not_missing[target_column]     # 非缺失值行的目标列\n\n        # 4. 初始化 LGBM 模型\n        if model_type == 'regression':\n            model = LGBMRegressor(n_estimators=n_estimators, random_state=random_state)\n        elif model_type == 'classification':\n            model = LGBMClassifier(n_estimators=n_estimators, random_state=random_state)\n        else:\n            raise ValueError(\"model_type should be either 'regression' or 'classification'.\")\n\n        # 5. 训练模型\n        model.fit(X_train, y_train)\n\n        # 6. 使用模型预测缺失值\n        X_pred = df_missing[features_columns]\n        y_pred = model.predict(X_pred)\n\n        # 7. 用预测值填充缺失值部分\n        df.loc[df[target_column].isnull(), target_column] = y_pred\n\n        # 8. 将数据集重新拆分为原来的 train 和 test\n        train_filled = df[df['is_train'] == 1].drop(columns=['is_train'])\n        test_filled = df[df['is_train'] == 0].drop(columns=['is_train'])\n\n        return train_filled, test_filled\n\n\n    tianbu_cols = ['CGAS-CGAS_Score','Physical-BMI','Physical-Height','Physical-Weight', \\\n                   'Physical-Diastolic_BP','Physical-HeartRate','Physical-Systolic_BP', \\\n                   'SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday']\n\n    for col in tianbu_cols:\n        print(\"开始填补特征：\"+ col)\n        train, test = fill_missing_with_lgbm(train, test, col)\n","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    FGC_cols = [\n      'FGC-FGC_CU',\n      'FGC-FGC_CU_Zone',\n      'FGC-FGC_GSND',\n      'FGC-FGC_GSND_Zone',\n      'FGC-FGC_GSD',\n      'FGC-FGC_GSD_Zone',\n      'FGC-FGC_PU',\n      'FGC-FGC_PU_Zone',\n      'FGC-FGC_SRL',\n      'FGC-FGC_SRL_Zone',\n      'FGC-FGC_SRR',\n      'FGC-FGC_SRR_Zone',\n      'FGC-FGC_TL',\n      'FGC-FGC_TL_Zone'\n    ]\n    for col in FGC_cols:\n        print(\"开始填补特征：\"+ col)\n        train, test = fill_missing_with_lgbm(train, test, col)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    BIA = [\"BIA-BIA_TBW\",\"BIA-BIA_TBW\",\"BIA-BIA_DEE\",\"BIA-BIA_BMC\",\"BIA-BIA_Fat\", \"BIA-BIA_BMI\",]\n    for col in BIA:\n        print(\"开始填补特征：\"+ col)\n        train, test = fill_missing_with_lgbm(train, test, col)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    train = update(train)\n    test = update(test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    train.hist(figsize=(15, 10), bins=20, xlabelsize=8, ylabelsize=8)\n\n    plt.tight_layout()\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    train = train.dropna(subset='sii')\n    train[\"sii\"].hist()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:    \n    \n    missing_percent_train = train.isnull().mean() * 100\n    missing_percent_train","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    missing_percent_test = test.isnull().mean() * 100\n    missing_percent_test","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    msno.matrix(train)\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    msno.matrix(test)\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    def create_interaction_features(df, feature_pairs):\n        global cat_c\n        for feature1, feature2 in feature_pairs:\n            if(feature1 not in cat_c or feature2 not in cat_c):\n                print(\"feature1:\" + feature1 + \",feature2:\" + feature2)\n                new_feature_name = f\"{feature1}_x_{feature2}\"\n                df[new_feature_name] = df[feature1] * df[feature2]\n        return df\n\n    def create_chufa_features(df, feature_pairs):\n        global cat_c\n        for feature1, feature2 in feature_pairs:\n            if feature1 not in cat_c or feature2 not in cat_c:\n                print(f\"feature1: {feature1}, feature2: {feature2}\")\n\n                # 构造 A/B 特征，先进行除法运算，但在除法之前控制出现0或NaN的情况\n                new_feature_name1 = f\"{feature1}_div_{feature2}\"\n                df[new_feature_name1] = df[feature1] / df[feature2]\n\n                # 当A或B中有0或NaN时保留NaN\n                df[new_feature_name1] = df[new_feature_name1].mask((df[feature1] == 0) | (df[feature2] == 0) | \n                                                                   df[feature1].isna() | df[feature2].isna())\n\n                # 构造 B/A 特征，类似方式处理\n                new_feature_name2 = f\"{feature2}_div_{feature1}\"\n                df[new_feature_name2] = df[feature2] / df[feature1]\n                df[new_feature_name2] = df[new_feature_name2].mask((df[feature1] == 0) | (df[feature2] == 0) | \n                                                                   df[feature1].isna() | df[feature2].isna())\n\n        return df","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    feature_pairs = [\n        ('PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age'),\n        ('Basic_Demos-Age', 'SDS-SDS_Total_T'),\n        ('FGC-FGC_SRR_Zone', 'SDS-SDS_Total_T'),\n        ('BIA-BIA_BMC', 'Physical-HeartRate'),\n        #('Fitness_Endurance-Season', 'Physical-Waist_Circumference'),\n        ('BIA-BIA_Fat', 'Physical-BMI'),\n        ('PreInt_EduHx-Season', 'Fitness_Endurance-Season'),\n        ('SDS-SDS_Total_T', 'Physical-Systolic_BP'),\n        ('Basic_Demos-Sex', 'FGC-FGC_PU_Zone')\n    ]\n\n\n    train = create_interaction_features(train, feature_pairs)\n    test = create_interaction_features(test, feature_pairs)\n    train = create_chufa_features(train, feature_pairs)\n    test = create_chufa_features(test, feature_pairs)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:    \n    \n    train","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    # モデルを学習した後のコード\n    XGBoost = xgb.XGBRegressor(random_state=SEED,enable_categorical=True)\n    XGBoost.fit(X, y)\n\n    # 特徴量重要度を取得\n    importance = XGBoost.feature_importances_\n\n    # 特徴量の名前を取得\n    features = X.columns\n\n    # データフレームとして整理\n    importance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\n\n    # 特徴量重要度を降順に並び替え\n    importance_df = importance_df.sort_values(by='Importance', ascending=False)\n\n    # プロット\n    plt.figure(figsize=(10, 20))\n    plt.barh(importance_df['Feature'], importance_df['Importance'])\n    plt.xlabel('Feature Importance')\n    plt.ylabel('Features')\n    plt.title('Feature Importance in XGBoost')\n    plt.gca().invert_yaxis()  # 重要度が高いものを上に\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    # 特徴量の名前を取得\n    features = X.columns\n\n    # データフレームとして整理（特徴量重要度と欠損値の割合を結合）\n    importance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\n    missing_df = pd.DataFrame({'Feature': missing_percent_train.index, 'MissingPercent': missing_percent_train.values})\n    combined_df = pd.merge(importance_df, missing_df, on='Feature')\n\n    # 散布図を作成\n    plt.figure(figsize=(10, 6))\n    plt.scatter(combined_df['MissingPercent'], combined_df['Importance'], alpha=0.7)\n    plt.xlabel('Missing Percentage (%)')\n    plt.ylabel('Feature Importance')\n    plt.title('Feature Importance vs Missing Percentage')\n    plt.grid(True)\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    # 获取各自列的集合\n    train_columns = set(train.columns)\n    test_columns = set(test.columns)\n\n    # 查找只在 train 中但不在 test 中的列\n    train_only = train_columns - test_columns\n\n    # 查找只在 test 中但不在 train 中的列\n    test_only = test_columns - train_columns\n\n    # 输出差异列\n    print(f\"Columns only in train: {train_only}\")\n    print(f\"Columns only in test: {test_only}\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n    def threshold_Rounder(oof_non_rounded, thresholds):\n        return np.where(oof_non_rounded < thresholds[0], 0,\n                        np.where(oof_non_rounded < thresholds[1], 1,\n                                 np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n    def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n        rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_p)\n\n    def TrainML(model_class, test_data):\n\n        X = train.drop(['sii'], axis=1)\n        y = train['sii']\n        test_data = test_data.drop(['sii'], axis=1)\n        SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n        train_S = []\n        test_S = []\n\n        oof_non_rounded = np.zeros(len(y), dtype=float) \n        oof_rounded = np.zeros(len(y), dtype=int) \n        test_preds = np.zeros((len(test_data), n_splits))\n\n        for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n            model = clone(model_class)\n            model.fit(X_train, y_train)\n\n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n\n            oof_non_rounded[test_idx] = y_val_pred\n            y_val_pred_rounded = y_val_pred.round(0).astype(int)\n            oof_rounded[test_idx] = y_val_pred_rounded\n\n            train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n            val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n            train_S.append(train_kappa)\n            test_S.append(val_kappa)\n\n            test_preds[:, fold] = model.predict(test_data)\n\n            print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            clear_output(wait=True)\n\n        print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n        print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n        KappaOPtimizer = minimize(evaluate_predictions,\n                                  x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                                  method='Nelder-Mead') # Nelder-Mead | # Powell\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n        tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n        print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n        submission = pd.DataFrame({\n            'id': sample['id'],\n            'sii': tpTuned\n        })\n\n        return submission, tKappa","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    best_params_lgbm = {'learning_rate': 0.011747572224219955, 'n_estimators': 993,  'min_child_weight': 0.0025036281384857462, 'colsample_bytree': 0.6648896193058901, 'reg_alpha': 0.7153672744430527, 'reg_lambda': 0.12158717311465662}\n    best_params_xgb = {'learning_rate': 0.007356059931165658, 'n_estimators': 957, 'subsample': 0.6555266544650088, 'colsample_bytree': 0.7712019245727745}\n    best_params_catboost = {'iterations': 804, 'learning_rate': 0.007849710402582562, 'l2_leaf_reg': 7.31183636902306, 'subsample': 0.5630297785016092, 'random_strength': 1.7097065892440113, 'bagging_temperature': 0.026593521316435192, 'border_count': 12}","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    # LightGBM\n    Light = lgb.LGBMRegressor(**best_params_lgbm, random_state=SEED, verbose=-1)\n    Submission_LGBM, k_lgbm = TrainML(Light, test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    # XGBoost\n    XGBoost = xgb.XGBRegressor(**best_params_xgb, random_state=SEED,enable_categorical=True)\n    Submission_XGB, k_xgb = TrainML(XGBoost, test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    # CatBoost\n    CatBoost = CatBoostRegressor(**best_params_catboost, random_state=SEED, verbose=0,cat_features=cat_c)\n    Submission_CatBoost , k_cat= TrainML(CatBoost, test)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    print(Submission_LGBM['sii'].value_counts())\n    print(Submission_XGB['sii'].value_counts())\n    print(Submission_CatBoost['sii'].value_counts())","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    # k_xgb = 0.6\n    # k_cat = 0.2\n    # k_lgbm = 0.2\n    # k値の合計を計算して、各モデルの重みを計算\n    total_k = k_cat + k_xgb + k_lgbm\n\n    weight_cat = k_cat / total_k    # CatBoost の重み\n    weight_xgb = k_xgb / total_k    # XGBoost の重み\n    weight_lgbm = k_lgbm / total_k  # LightGBM の重み\n\n    # 各モデルの予測結果（submission）を用意\n    # 'sii' がカテゴリラベルであることを前提とします\n    ensemble_df = pd.DataFrame({\n        'id': Submission_LGBM['id'],\n        'cat': Submission_CatBoost[\"sii\"],\n        'xgb': Submission_XGB[\"sii\"],\n        'lgbm': Submission_LGBM[\"sii\"]\n    })\n\n    # 予測結果を長い形式に変換\n    melted = ensemble_df.melt(id_vars='id', value_vars=['cat', 'xgb', 'lgbm'], \n                              var_name='model', value_name='sii')\n\n    # 各モデルに対応する重みを割り当て\n    melted['weight'] = melted['model'].map({\n        'cat': weight_cat,\n        'xgb': weight_xgb,\n        'lgbm': weight_lgbm\n    })\n\n    # 各idごと、siiごとに重みを集計\n    grouped = melted.groupby(['id', 'sii'])['weight'].sum().reset_index()\n\n    # 各idごとに最大の重みを持つsiiを選択\n    best_submission = grouped.loc[grouped.groupby('id')['weight'].idxmax()][['id', 'sii']]","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    comparison_df = best_submission.merge(ensemble_df, on='id', how='left')","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n\n    comparison_df","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'SOLUTION_17' in ENSEMBLE_SOLUTIONS:\n    \n    best_submission.to_csv('submission_17.csv', index=False)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ensemble solution","metadata":{}},{"cell_type":"code","source":"    \nif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 52':   \n    \n    sm11 = pd.read_csv('submission_11.csv')\n    sm14 = pd.read_csv('submission_14.csv')\n    sm15 = pd.read_csv('submission_15.csv')\n    sm11 = sm11.rename(columns={'sii': 'sii_11'})\n    sm14 = sm14.rename(columns={'sii': 'sii_14'})\n    sm15 = sm15.rename(columns={'sii': 'sii_15'})\n    sms  = pd.merge(sm11,sm14,on=['id'])\n    sms  = pd.merge(sms, sm15,on=['id'])\n    display(sms)\n    sms['sii'] = np.round(sms['sii_11'] *0.460 + 0.460* sms['sii_14'] + 0.080* sms['sii_15']).astype(int)\n    sms['sii'] = sms['sii'].astype(int)\n    \nelif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 60':   \n    \n    sm11 = pd.read_csv('submission_11.csv')\n    sm14 = pd.read_csv('submission_14.csv')\n    sm15 = pd.read_csv('submission_15.csv')\n    sm11 = sm11.rename(columns={'sii': 'sii_11'})\n    sm14 = sm14.rename(columns={'sii': 'sii_14'})\n    sm15 = sm15.rename(columns={'sii': 'sii_15'})\n    sms  = pd.merge(sm11,sm14,on=['id'])\n    sms  = pd.merge(sms, sm15,on=['id'])\n    display(sms)\n    sms['sii'] = np.round(sms['sii_11'] *0.460 + 0.460* sms['sii_14'] + 0.080* sms['sii_15']).astype(int)\n    sms['sii'] = sms['sii'].astype(int)\n    \nelif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 61':   \n    \n    sm11 = pd.read_csv('submission_11.csv')\n    sm14 = pd.read_csv('submission_14.csv')\n    sm15 = pd.read_csv('submission_15.csv')\n    sm11 = sm11.rename(columns={'sii': 'sii_11'})\n    sm14 = sm14.rename(columns={'sii': 'sii_14'})\n    sm15 = sm15.rename(columns={'sii': 'sii_15'})\n    sms  = pd.merge(sm11,sm14,on=['id'])\n    sms  = pd.merge(sms, sm15,on=['id'])\n    display(sms)\n    sms['sii'] = np.round(sms['sii_11'] *0.460 + 0.460* sms['sii_14'] + 0.080* sms['sii_15']).astype(int)\n    sms['sii'] = sms['sii'].astype(int)\n    \nelif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 62':   \n    \n    sm11 = pd.read_csv('submission_11.csv')\n    sm14 = pd.read_csv('submission_14.csv')\n    sm15 = pd.read_csv('submission_15.csv')\n    sm11 = sm11.rename(columns={'sii': 'sii_11'})\n    sm14 = sm14.rename(columns={'sii': 'sii_14'})\n    sm15 = sm15.rename(columns={'sii': 'sii_15'})\n    sms  = pd.merge(sm11,sm14,on=['id'])\n    sms  = pd.merge(sms, sm15,on=['id'])\n    display(sms)\n    sms['sii'] = np.round(sms['sii_11'] *0.460 + 0.460* sms['sii_14'] + 0.080* sms['sii_15']).astype(int)\n    sms['sii'] = sms['sii'].astype(int)\n","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = sms[['id','sii']]\nsub.to_csv('submission.csv', index=False)\nsub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # arhiv\n\n# if ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_16','SOLUTION_15'] and LAUNCH_VARIANT == 'option 59':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm16 = pd.read_csv('submission_16.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm16 = sm16.rename(columns={'sii': 'sii_16'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm16,on=['id'])\n#     sms  = pd.merge(sms, sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.460 + 0.460* sms['sii_16'] + 0.080* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 43':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 50':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     sms  = pd.merge(sms, sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.425 + 0.425* sms['sii_14'] + 0.150* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 51':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     sms  = pd.merge(sms, sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.440 + 0.440* sms['sii_14'] + 0.120* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 52':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     sms  = pd.merge(sms, sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.460 + 0.460* sms['sii_14'] + 0.080* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 53':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     sms  = pd.merge(sms, sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.455 + 0.455* sms['sii_14'] + 0.090* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 54':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     sms  = pd.merge(sms, sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.465 + 0.465* sms['sii_14'] + 0.070* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 55':   \n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm14,sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_14'] *0.50 + 0.50* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_14','SOLUTION_16'] and LAUNCH_VARIANT == 'option 56':   \n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm16 = pd.read_csv('submission_16.csv')\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm16 = sm16.rename(columns={'sii': 'sii_16'})\n#     sms  = pd.merge(sm14,sm16,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_14'] *0.50 + 0.50* sms['sii_16']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_15','SOLUTION_16'] and LAUNCH_VARIANT == 'option 57':   \n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm16 = pd.read_csv('submission_16.csv')\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sm16 = sm16.rename(columns={'sii': 'sii_16'})\n#     sms  = pd.merge(sm15,sm16,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_15'] *0.50 + 0.50* sms['sii_16']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_14','SOLUTION_15','SOLUTION_16'] and LAUNCH_VARIANT == 'option 58':   \n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm16 = pd.read_csv('submission_16.csv')\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sm16 = sm16.rename(columns={'sii': 'sii_16'})\n#     sms  = pd.merge(sm14,sm15,on=['id'])\n#     sms  = pd.merge(sms, sm16,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_14'] *0.333 + 0.334* sms['sii_15'] + 0.333* sms['sii_16']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n    \n    \n    \n# - option 55->Lb=0.46? solutions (14,15)\n# - option 56->Lb=0.46? solutions (14,16)\n# - option 57->Lb=0.46? solutions (15,16)\n# - option 58->Lb=0.46? solutions (14,15,16)\n\n# next option\n# LAUNCH_VARIANT,ENSEMBLE_SOLUTIONS = 'option 55',['SOLUTION_14','SOLUTION_15']\n# LAUNCH_VARIANT,ENSEMBLE_SOLUTIONS = 'option 56',['SOLUTION_14','SOLUTION_16']\n# LAUNCH_VARIANT,ENSEMBLE_SOLUTIONS = 'option 57',['SOLUTION_15','SOLUTION_16']\n# LAUNCH_VARIANT,ENSEMBLE_SOLUTIONS = 'option 58',['SOLUTION_14','SOLUTION_15','SOLUTION_16']\n\n\n\n\n# if ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 48':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     sms  = pd.merge(sms, sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.474 + 0.474* sms['sii_14'] + 0.052* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# if ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14','SOLUTION_15'] and LAUNCH_VARIANT == 'option 49':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     sms  = pd.merge(sms, sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.450 + 0.450* sms['sii_14'] + 0.100* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# if ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 43':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_15'] and LAUNCH_VARIANT == 'option 45':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_15'] and LAUNCH_VARIANT == 'option 46':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_15'] and LAUNCH_VARIANT == 'option 47':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm15 = pd.read_csv('submission_15.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm15 = sm15.rename(columns={'sii': 'sii_15'})\n#     sms  = pd.merge(sm11,sm15,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_15']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# if ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 41':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 42':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 43':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# if ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 37':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 38':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 39':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 40':   \n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n    \n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_11','SOLUTION_14'] and LAUNCH_VARIANT == 'option 36':\n    \n#     sm11 = pd.read_csv('submission_11.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm11 = sm11.rename(columns={'sii': 'sii_11'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm11,sm14,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_11'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_10','SOLUTION_13']:\n    \n#     sm10 = pd.read_csv('submission_10.csv')\n#     sm13 = pd.read_csv('submission_13.csv')\n#     sm10 = sm10.rename(columns={'sii': 'sii_10'})\n#     sm13 = sm13.rename(columns={'sii': 'sii_13'})\n#     sms  = pd.merge(sm10,sm13,on=['id'])\n#     display(sms)\n#     sms['sii'] = np.round(sms['sii_10'] *0.50 + 0.50* sms['sii_13']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# if ENSEMBLE_SOLUTIONS == ['SOLUTION_10','SOLUTION_13'] and LAUNCH_VARIANT == 'option 31':\n    \n#     sm10 = pd.read_csv('submission_10.csv')\n#     sm13 = pd.read_csv('submission_13.csv')\n#     sm10 = sm10.rename(columns={'sii': 'sii_10'})\n#     sm13 = sm13.rename(columns={'sii': 'sii_13'})\n#     sms  = pd.merge(sm10,sm13,on=['id'])\n#     display(sms,sm10,sm13)\n#     sms['sii'] = np.round(sms['sii_10'] *0.6 + 0.4* sms['sii_13']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_10','SOLUTION_13'] and LAUNCH_VARIANT == 'option 32':\n    \n#     sm10 = pd.read_csv('submission_10.csv')\n#     sm13 = pd.read_csv('submission_13.csv')\n#     sm10 = sm10.rename(columns={'sii': 'sii_10'})\n#     sm13 = sm13.rename(columns={'sii': 'sii_13'})\n#     sms  = pd.merge(sm10,sm13,on=['id'])\n#     display(sms,sm10,sm13)\n#     sms['sii'] = np.round(sms['sii_10'] *0.4 + 0.6* sms['sii_13']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_10','SOLUTION_13'] and LAUNCH_VARIANT == 'option 33':\n    \n#     sm10 = pd.read_csv('submission_10.csv')\n#     sm13 = pd.read_csv('submission_13.csv')\n#     sm10 = sm10.rename(columns={'sii': 'sii_10'})\n#     sm13 = sm13.rename(columns={'sii': 'sii_13'})\n#     sms  = pd.merge(sm10,sm13,on=['id'])\n#     display(sms,sm10,sm13)\n#     sms['sii'] = np.round(sms['sii_10'] *0.55 + 0.45* sms['sii_13']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_10','SOLUTION_13'] and LAUNCH_VARIANT == 'option 34':\n    \n#     sm10 = pd.read_csv('submission_10.csv')\n#     sm13 = pd.read_csv('submission_13.csv')\n#     sm10 = sm10.rename(columns={'sii': 'sii_10'})\n#     sm13 = sm13.rename(columns={'sii': 'sii_13'})\n#     sms  = pd.merge(sm10,sm13,on=['id'])\n#     display(sms,sm10,sm13)\n#     sms['sii'] = np.round(sms['sii_10'] *0.45 + 0.55* sms['sii_13']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_10','SOLUTION_13'] and LAUNCH_VARIANT == 'option 35':\n    \n#     sm10 = pd.read_csv('submission_10.csv')\n#     sm13 = pd.read_csv('submission_13.csv')\n#     sm10 = sm10.rename(columns={'sii': 'sii_10'})\n#     sm13 = sm13.rename(columns={'sii': 'sii_13'})\n#     sms  = pd.merge(sm10,sm13,on=['id'])\n#     display(sms,sm10,sm13)\n#     sms['sii'] = np.round(sms['sii_10'] *0.49 + 0.51* sms['sii_13']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n    \n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_10','SOLUTION_13']:\n    \n#     sm10 = pd.read_csv('submission_10.csv')\n#     sm13 = pd.read_csv('submission_13.csv')\n#     sm10 = sm10.rename(columns={'sii': 'sii_10'})\n#     sm13 = sm13.rename(columns={'sii': 'sii_13'})\n#     sms  = pd.merge(sm10,sm13,on=['id'])\n#     display(sms,sm10,sm13)\n#     sms['sii'] = np.round(sms['sii_10'] *0.50 + 0.50* sms['sii_13']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_10','SOLUTION_14']:\n    \n#     sm10 = pd.read_csv('submission_10.csv')\n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm10 = sm10.rename(columns={'sii': 'sii_10'})\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sms  = pd.merge(sm10,sm14,on=['id'])\n#     display(sms,sm10,sm14)\n#     sms['sii'] = np.round(sms['sii_10'] *0.50 + 0.50* sms['sii_14']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_13','SOLUTION_14']:\n    \n#     sm14 = pd.read_csv('submission_14.csv')\n#     sm13 = pd.read_csv('submission_13.csv')\n#     sm14 = sm14.rename(columns={'sii': 'sii_14'})\n#     sm13 = sm13.rename(columns={'sii': 'sii_13'})\n#     sms  = pd.merge(sm14,sm13,on=['id'])\n#     display(sms,sm14,sm13)\n#     sms['sii'] = np.round(sms['sii_14'] *0.50 + 0.50* sms['sii_13']).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# elif ENSEMBLE_SOLUTIONS == ['SOLUTION_1','SOLUTION_10','SOLUTION_11']:\n    \n#     sm1 = pd.read_csv('submission_1.csv')\n#     sm11= pd.read_csv('submission_11.csv')\n#     sm10= pd.read_csv('submission_10.csv')\n#     sm1 = sm1 .rename(columns={'sii': 'sii_1'})\n#     sm11= sm11.rename(columns={'sii': 'sii_11'})\n#     sm10= sm10.rename(columns={'sii': 'sii_10'})\n#     sms = pd.merge(sm1,sm11,on=['id'])\n#     sms = pd.merge(sms,sm10,on=['id'])\n#     display(sms,sm10,sm11,sm1)\n#     sms['sii'] = np.round((sms['sii_10'] + sms['sii_11'] + sms['sii_10'])/3).astype(int)\n#     sms['sii'] = sms['sii'].astype(int)\n\n# # if ENSEMBLE_SOLUTIONS == ['SOLUTION_8','SOLUTION_10']:\n    \n# #     sm8 = pd.read_csv('submission_8.csv')\n# #     sm10= pd.read_csv('submission_10.csv')\n# #     sm8 = sm8 .rename(columns={'sii': 'sii_8'})\n# #     sm10= sm10.rename(columns={'sii': 'sii_10'})\n# #     sms = pd.merge(sm8,sm10,on=['id'])\n# #     display(sms,sm8,sm10)\n# #     sms['sii'] = np.round(sms['sii_8'] *0.45 + 0.55* sms['sii_10']).astype(int)\n# #     sms['sii'] = sms['sii'].astype(int)\n\n# # sm1 = pd.read_csv('submission_1.csv')\n# # sm8 = pd.read_csv('submission_8.csv')\n# # sm2 = pd.read_csv('submission_2.csv')\n# # display(sm1,sm2,sm8)\n# # sm1 = sm1.rename(columns={'sii': 'sii_1'})\n# # sm8 = sm8.rename(columns={'sii': 'sii_8'})\n# # sm2 = sm2.rename(columns={'sii': 'sii_2'})\n# # sms = pd.merge(sm1,sm8, on=['id'])\n# # sms = pd.merge(sms,sm2, on=['id'])\n\n# # sms['sii'] = np.round(sms['sii_1']*0.34+sms['sii_8']*0.35+sms['sii_2']*0.31).astype(int)   # 0.443\n\n# # sms['sii'] = sms['sii'].astype(int)\n\n# # ~ ~ ~ ~ ~ ~ ~ ~ ~ ~ ~ ~ ~ ~ ~ ~ ~ ~\n\n# # sm1 = pd.read_csv('submission_1.csv')\n# # sm8 = pd.read_csv('submission_8.csv')\n# # sm2 = pd.read_csv('submission_2.csv')\n# # display(sm1,sm2,sm8)\n# # sm1 = sm1.rename(columns={'sii': 'sii_1'})\n# # sm8 = sm8.rename(columns={'sii': 'sii_8'})\n# # sm2 = sm2.rename(columns={'sii': 'sii_2'})\n# # sms = pd.merge(sm1,sm8, on=['id'])\n# # sms = pd.merge(sms,sm2, on=['id'])\n\n# # sms['sii'] = np.round((sms['sii_1']+sms['sii_8']+sms['sii_2'])/3).astype(int)              # 0.4??\n\n# # sms['sii'] = sms['sii'].astype(int)\n\n# # some rezults\n# # - option 1 **<** ONODERA\n# # - option 3 **<** option 1\n# # - option 01->Lb=0.448 solutions (1,2,3) + weight(0.500+0.500+0.000)\n# # - option 03->Lb=0.448 solutions (1,2,3) + weight(0.333+0.333+0.333)\n# # - option 10->Lb=0.436 solutions (1,2,3) + weight(0.409+0.334+0.357)\n# # - option 11->Lb=0.427 solutions (1,2,3) + weight(0.800+0.150+0.050)\n# # - option 24.weight(0.01+0.99) < option 22.weight\n\n# # ? weight(= = =), Lb=**0.448** solutions (1,2,3) & Lb=**0.443** solutions (1,8,2)\n# # ? weight(= = =), Lb=**0.448**(0.450; 0.441; 0.345) & Lb=**0.443**(0.450; 0.452; 0.441)\n# # ? weight(= = =), solu.3(0.345) < solu.8(0.452) ! however, the ensemble with the solu.3 is alarger !\n\n# # - option 26->Lb=0.4?? solutions (1,8,2) weight(0.333+0.333+0.333)\n# # - option 12->Lb=0.4?? solutions (5,7) weight(50x50)\n# # - option 15->Lb=0.4?? solutions (7,8) weight(50x50)\n# # - option 16->Lb=0.4?? solutions (1,2) weight(60x40)\n# # - option 17->Lb=0.4?? solutions (1,2) weight(40x60)\n# # - ?\n# # - ?\n# # - option 25->Lb=0.45? solutions (1,8) weight(0.08+0.92)\n# # - ?\n# # - ?\n# # - option 18->Lb=0.4?? solutions (1,2,3) + weight(0.30+0.35+0.35)\n# # - option 19->Lb=0.4?? solutions (1,2,3) + weight(0.30+0.40+0.30)\n# # - option 20->Lb=0.4?? solutions (1,2,3) + weight(0.30+0.30+0.40)\n\n# # sm1 = pd.read_csv('submission_1.csv')\n# # sm2 = pd.read_csv('submission_2.csv')\n# # sm3 = pd.read_csv('submission_3.csv')\n# # display(sm1,sm2,sm3)\n# # sm1 = sm1.rename(columns={'sii': 'sii_1'})\n# # sm2 = sm2.rename(columns={'sii': 'sii_2'})\n# # sm3 = sm3.rename(columns={'sii': 'sii_3'})\n# # sms = pd.merge(sm1,sm2, on=['id'])\n# # sms = pd.merge(sms,sm3, on=['id'])\n\n# # # sms['sii'] = np.round(sms['sii_1']*0.800+sms['sii_2']*0.150+sms['sii_3']*0.050).astype(int) # 0.427\n# # # sms['sii'] = np.round(sms['sii_1']*0.409+sms['sii_2']*0.334+sms['sii_3']*0.357).astype(int) # 0.436\n# # # sms['sii'] = np.round(sms['sii_1']*0.5  +sms['sii_2']*0.3  +sms['sii_3']*0.200).astype(int) # 0.435\n# # # sms['sii'] = np.round(sms['sii_1']*0.334+sms['sii_2']*0.333+sms['sii_3']*0.333).astype(int) # 0.4481\n# # # sms['sii'] = np.round(sms['sii_1']*0.5  +sms['sii_2']*0.5  +sms['sii_3']*0.0  ).astype(int) # 0.4483\n\n# # sms['sii'] = sms['sii'].astype(int)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}