{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The second model from the notebook [Prem Chotepanit - CMI| Tuning | Ensemble of solutions 0.497](https://www.kaggle.com/code/batprem/cmi-tuning-ensemble-of-solutions)","metadata":{}},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-12T19:26:59.463919Z","iopub.execute_input":"2024-12-12T19:26:59.464173Z","iopub.status.idle":"2024-12-12T19:27:40.971923Z","shell.execute_reply.started":"2024-12-12T19:26:59.464146Z","shell.execute_reply":"2024-12-12T19:27:40.970751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport abc\nimport enum\nimport datetime\nimport time\nimport copy\nimport random\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom collections import OrderedDict, defaultdict\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom IPython import get_ipython\nfrom IPython.display import clear_output\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.optim import lr_scheduler\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nfrom pytorch_tabnet.callbacks import Callback\n\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\nfrom scipy.optimize import minimize\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import KFold, StratifiedKFold, train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\n\nfrom colorama import Fore, Back, Style\nb_ = Fore.BLUE\nsr_ = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\npd.options.display.max_columns = None\n#pd.options.display.max_rows = None\n\n@enum.unique\nclass RunModeEnum(enum.IntEnum):\n    Train = 0\n    Infer = 1\n\n@enum.unique\nclass DataEnum(enum.IntEnum):\n    Train = 0\n    Valid = 1\n    Test = 2\n    Infer = 3\n\nclass APP:\n    version = None\n    debug = False\n    run_mode = RunModeEnum.Train\n    short_dataset = False\n    test_full_dataset = False\n    check_inference = True  # False\n    three_phase = False\n    used_gateway_server = False\n    used_gpu = True    \n    gpu_float_64 = False # True    \n    parallel_gpu = False\n    disable_gpu_tf = False\n    disable_cpu_tf = False    \n    kaggle = os.environ.get(\"KAGGLE_KERNEL_RUN_TYPE\", \"\") != \"\"\n    submit = os.environ.get('KAGGLE_IS_COMPETITION_RERUN', \"\") != \"\"\n    local = os.environ.get(\"DOCKER_USING\", \"\") == \"LOCAL\"\n    try:\n        interactive = 'runtime' in get_ipython().config.IPKernelApp.connection_file\n    except Exception as inst:\n        print(\"Error interactive:\", inst)\n        interactive = False\n    jupyter = \"ipykernel\" in globals()\n    if not jupyter:\n        try:\n            if \"IPython\" in globals().get(\"__doc__\", \"\"):\n                jupyter = True\n        except Exception as inst:\n            print(\"Error IPython:\", inst)\n    if submit:\n        debug = False\n        short_dataset = False\n        test_full_dataset = False\n        check_inference = False\n        run_mode = RunModeEnum.Infer\n    if debug:\n        num_workers = 0\n        # For descriptive error messages\n        os.environ['CUDA_LAUNCH_BLOCKING'] = \"1\"\n        os.environ['TORCH_USE_CUDA_DSA'] = \"1\"\n        os.environ['MKL_NUM_THREADS'] = '1' \n        os.environ['OPENBLAS_NUM_THREADS'] = '1'\n        os.environ[\"NUM_INTER_THREADS\"] = \"1\"\n        os.environ[\"NUM_INTRA_THREADS\"] = \"1\"\n        os.environ[\"XLA_FLAGS\"] = (\"--xla_cpu_multi_thread_eigen=false \"\n                                   \"intra_op_parallelism_threads=1\")        \n        #tf.config.threading.set_inter_op_parallelism_threads(1)\n        #tf.config.threading.set_intra_op_parallelism_threads(1)\n    else:\n        num_workers = os.cpu_count()\n        #parallel_gpu = True\n    if kaggle:\n        #parallel_gpu = True\n        # GPU_DEVICES = \"auto\"\n        pass\n    device = torch.device(\"cuda\")\n    device_0 = torch.device(\"cuda:0\")\n    gpu_count = torch.cuda.device_count()\n    if parallel_gpu and gpu_count > 1:\n        num_gpu_process = gpu_count\n        device_1 = torch.device(\"cuda:1\")\n        os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1\"\n        num_workers //= 2\n    else:\n        num_gpu_process = 1\n        device_1 = device_0\n        os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0\"\n    if debug:\n        print(f\"{Back.CYAN}mode: DEBUG!{sr_}\")\n    print(f\"jupyter:{jupyter}, kaggle:{kaggle}, local:{local}, submit:{submit}, interactive:{interactive}, gpu_count:{gpu_count}\")\n    if torch.cuda.is_available():\n        print(\"[INFO] Using GPU: {}\\n\".format(torch.cuda.get_device_name()))\n    do_cross_val_score = not submit\n\n    date_time_start = datetime.datetime.now()\n    dt_start_ymd_hms = date_time_start.strftime(\"%Y.%m.%d_%H-%M-%S\")\n\n    file_run_path = Path(\"\")\n    if jupyter:\n        try:\n            file_run_path = Path(globals().get(\"__vsc_ipynb_file__\", \"\"))\n            print(f\"file_run_path globals:{file_run_path}\")\n        except Exception as inst:\n            print('file_run_path globals:',inst)\n    else:\n        try:\n            file_run_path = Path(__file__)\n            print(f\"file_run_path:{file_run_path}\")\n        except Exception as inst:\n            print('file_run_path:',inst)\n    file_run_name = file_run_path.stem\n    if version is None:\n        version = file_run_name.split(' ')[0]\n    path_app = file_run_path.parent\n    path_run = Path(os.getcwd())\n    log_dir = (version + \"_\" if version.strip() else \"\") + \"weights_\" + dt_start_ymd_hms\n    path_out = Path(\"/kaggle/working\") if kaggle else path_app / log_dir\n    output_dir = \"/kaggle/working\" if kaggle else \".\"\n    if not os.path.exists(path_out):\n        os.makedirs(path_out)\n    path_log = f\"{output_dir}/{log_dir}\"\n    if not os.path.exists(path_log):\n        os.makedirs(path_log)\n    path_model = f\"{output_dir}/{log_dir}\"\n    if not os.path.exists(path_model):\n        os.makedirs(path_model)\n    path_root = Path('/kaggle/input')\n    print(f\"path_app: {path_app}\")\n    print(f\"path_run: {path_run}\")\n    print(f\"log_dir: {log_dir}\")\n    print(f\"path_out: {path_out}\")\n    print(f\"path_log: {path_log}\")\n    print(f\"path_model: {path_model}\")\n\n!cat /etc/os-release | grep -oP \"PRETTY_NAME=\\\"\\K([^\\\"]*)\" && uname -r\nprint(f\"CONTAINER_NAME={os.environ['CONTAINER_NAME']}, BUILD_DATE={os.environ['BUILD_DATE']}\")\n!free -h\n!nv_version=\"$(nvidia-smi --query-gpu=driver_version --format=csv,noheader)\" && echo \"My NVIDIA driver version is '${nv_version}'.\"\n!ls -l /usr/local | grep cuda\n\nSEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T19:27:40.974523Z","iopub.execute_input":"2024-12-12T19:27:40.974902Z","iopub.status.idle":"2024-12-12T19:27:51.933880Z","shell.execute_reply.started":"2024-12-12T19:27:40.974860Z","shell.execute_reply":"2024-12-12T19:27:51.932939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        X_imputed = self.imputer.fit_transform(X)\n        if hasattr(y, 'values'):\n            y = y.values\n        X_train, X_valid, y_train, y_valid = train_test_split(X_imputed, y, test_size=0.2, random_state=SEED)\n        history = self.model.fit(\n            X_train = X_train,\n            y_train = y_train.reshape(-1, 1),\n            eval_set = [(X_valid, y_valid.reshape(-1, 1))],\n            eval_name = ['valid'],\n            eval_metric = ['mse'],\n            max_epochs = 10 if APP.short_dataset else 200,\n            patience = 20,\n            batch_size = 1024,\n            virtual_batch_size = 128,\n            num_workers = 0,\n            drop_last = False,\n            callbacks = [\n                TabNetPretrainedModelCheckpoint(\n                    filepath = self.best_model_path,\n                    monitor = 'valid_mse',\n                    mode = 'min',\n                    save_best_only = True,\n                    verbose = True\n                )\n            ]\n        )\n\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)\n        return self\n\n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n\n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = self.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', save_best_only=True, verbose=1):\n        super().__init__()\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        if (self.mode == 'min' and current < self.best) or (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T19:27:51.935208Z","iopub.execute_input":"2024-12-12T19:27:51.936272Z","iopub.status.idle":"2024-12-12T19:27:51.950061Z","shell.execute_reply.started":"2024-12-12T19:27:51.936226Z","shell.execute_reply":"2024-12-12T19:27:51.949263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    train_S = []\n    test_S = []\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        test_preds[:, fold] = model.predict(test_data)\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)\n#ODT_Model = ObliqueDecisionTreeRegressor(**ODT_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    #('tabnet', TabNet_Model),\n    #('odt', ODT_Model),\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n# Save submission\nSubmission2.to_csv('submission.csv', index=False)\ndisplay(Submission2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T19:27:51.952192Z","iopub.execute_input":"2024-12-12T19:27:51.952554Z","iopub.status.idle":"2024-12-12T19:29:50.917033Z","shell.execute_reply.started":"2024-12-12T19:27:51.952509Z","shell.execute_reply":"2024-12-12T19:29:50.916132Z"}},"outputs":[],"execution_count":null}]}