{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Child Mind Institute — Problematic Internet Use: Modeling and Evaluation\n\n## Objective\nDevelop machine learning models to predict problematic internet use (PIU) in children based on provided survey data. The task involves leveraging feature engineering and advanced predictive techniques to achieve accurate classification. Evaluation will prioritize Quadratic Weighted Kappa (QWK), which measures agreement between predicted and actual labels, accounting for ordinal classification nuances. The goal is to create a robust and generalizable model that supports efforts to better understand and address PIU in children.\n\n## Outline\n1. Import Libraries\n2. Data Loading and Processing\n3. Metrics\n4. Modeling and Evaluation\n5. Hypertuning\n6. Model 1\n7. Model 2\n8. Encoding & Feature Engineering\n9. Model 3\n10. Ensemble & Submission","metadata":{}},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:44:01.448391Z","iopub.execute_input":"2024-12-19T05:44:01.448724Z","iopub.status.idle":"2024-12-19T05:44:42.930978Z","shell.execute_reply.started":"2024-12-19T05:44:01.448680Z","shell.execute_reply":"2024-12-19T05:44:42.929826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\n\nfrom scipy.optimize import minimize\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import mean_squared_error, accuracy_score, cohen_kappa_score\n\nimport optuna\nfrom sklearn.pipeline import Pipeline\nfrom lightgbm import LGBMRegressor, early_stopping\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nfrom sklearn.base import BaseEstimator, RegressorMixin, clone\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor, VotingRegressor\n\nimport torch\nimport torch.nn as nn\nfrom tqdm import tqdm\nfrom tqdm import tqdm_notebook\nimport torch.optim as optim\nfrom pytorch_tabnet.callbacks import Callback\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:44:42.932868Z","iopub.execute_input":"2024-12-19T05:44:42.933176Z","iopub.status.idle":"2024-12-19T05:44:49.798719Z","shell.execute_reply.started":"2024-12-19T05:44:42.933146Z","shell.execute_reply":"2024-12-19T05:44:49.797796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:44:49.799806Z","iopub.execute_input":"2024-12-19T05:44:49.800341Z","iopub.status.idle":"2024-12-19T05:44:49.804179Z","shell.execute_reply.started":"2024-12-19T05:44:49.800313Z","shell.execute_reply":"2024-12-19T05:44:49.803273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Loading and Processing","metadata":{}},{"cell_type":"code","source":"# Load data\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Store feature names (test specific)\nfeats = test_df.columns.tolist()\nfeats.remove('id')\nfeats.append('sii')\n\n# Display initial rows\ntrain_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:44:49.806192Z","iopub.execute_input":"2024-12-19T05:44:49.806472Z","iopub.status.idle":"2024-12-19T05:44:49.902558Z","shell.execute_reply.started":"2024-12-19T05:44:49.806421Z","shell.execute_reply":"2024-12-19T05:44:49.901677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Series parquet process\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    \n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n# Process accelerometer series\ndef load_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(\n            lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:44:49.903398Z","iopub.execute_input":"2024-12-19T05:44:49.903653Z","iopub.status.idle":"2024-12-19T05:44:49.909721Z","shell.execute_reply.started":"2024-12-19T05:44:49.903628Z","shell.execute_reply":"2024-12-19T05:44:49.908808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s_train = load_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ns_test = load_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# Merge series with tabular data\ntrain = pd.merge(train_df, s_train, how=\"left\", on='id')\ntest = pd.merge(test_df, s_test, how=\"left\", on='id')\n\n# Drop id column\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\n# Store series column names w/ test features\nseries_cols = s_train.columns.tolist()\nseries_cols.remove(\"id\")\nfeatures = feats + series_cols\n\n# Keep useful training data\ntrain = train[features]\ntrain = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:44:49.910759Z","iopub.execute_input":"2024-12-19T05:44:49.911026Z","iopub.status.idle":"2024-12-19T05:46:05.177399Z","shell.execute_reply.started":"2024-12-19T05:44:49.911001Z","shell.execute_reply":"2024-12-19T05:46:05.176495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store categorical columns\ncat_cols = train.select_dtypes(include=[\"object\"]).columns.tolist()\n\n# Filter null categorical columns\ndef filter_cat(df):\n    global cat_cols\n    for c in cat_cols: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n        \n    return df\n        \ntrain = filter_cat(train)\ntest = filter_cat(test)\n\n# Create unique label dictionary\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    \n    return {value: idx for idx, value in enumerate(unique_values)}\n\n# Label encoding for categorical columns\nfor col in cat_cols:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.178599Z","iopub.execute_input":"2024-12-19T05:46:05.178891Z","iopub.status.idle":"2024-12-19T05:46:05.241285Z","shell.execute_reply.started":"2024-12-19T05:46:05.178865Z","shell.execute_reply":"2024-12-19T05:46:05.240265Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Metrics","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    \n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.242518Z","iopub.execute_input":"2024-12-19T05:46:05.242922Z","iopub.status.idle":"2024-12-19T05:46:05.251670Z","shell.execute_reply.started":"2024-12-19T05:46:05.242871Z","shell.execute_reply":"2024-12-19T05:46:05.250136Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Modeling and Evaluation","metadata":{}},{"cell_type":"code","source":"# Tabnet wrapper for classification\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        # Initialize TabNet model, imputer, and model path for saving best model\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values and split data into train/validation\n        X_imputed = self.imputer.fit_transform(X)\n        X_train, X_valid, y_train, y_valid = train_test_split(X_imputed, y, test_size=0.2, random_state=42)\n        \n        # Train the TabNet model with validation\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=500,\n            patience=50,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[TabNetPretrainedModelCheckpoint(filepath=self.best_model_path, monitor='valid_mse', mode='min', save_best_only=True, verbose=True)]\n        )\n        \n        # Load best model after training\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Clean up temporary file\n        \n        return self\n    \n    def predict(self, X):\n        # Impute missing values and make predictions\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        # Add deepcopy support for scikit-learn compatibility\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        # Initialize model reference for saving\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        # Track improvements in monitored metric and save model if best\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Compare current metric with best value\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)  # Save model if improvement","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.253006Z","iopub.execute_input":"2024-12-19T05:46:05.253351Z","iopub.status.idle":"2024-12-19T05:46:05.268407Z","shell.execute_reply.started":"2024-12-19T05:46:05.253314Z","shell.execute_reply":"2024-12-19T05:46:05.267354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model\ndef TrainML(model_class, train, test_data):\n    # Prepare training features and target variable\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    # Initialize stratified K-fold cross-validation\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    # Initialize placeholders for metrics and predictions\n    train_S = []\n    test_S = []\n    oof_non_rounded = np.zeros(len(y), dtype=float)  # Out-of-fold predictions\n    oof_rounded = np.zeros(len(y), dtype=int)  # Rounded OOF predictions\n    test_preds = np.zeros((len(test_data), n_splits))  # Test set predictions\n\n    # Perform cross-validation\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        # Split data into training and validation sets\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        # Train a cloned model on the training set\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        # Predict on train and validation sets\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        # Store out-of-fold predictions\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)  # Round validation predictions\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        # Compute quadratic weighted kappa for training and validation\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        # Store metrics for each fold\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        # Predict on test set for this fold\n        test_preds[:, fold] = model.predict(test_data)\n        \n        # Print fold results\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    # Display mean metrics across all folds\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Optimize thresholds for rounding predictions\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    # Apply optimized thresholds to out-of-fold predictions\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    # Print optimized quadratic weighted kappa score\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    # Average test predictions across folds and apply thresholds\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    # Prepare submission dataframe\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.273275Z","iopub.execute_input":"2024-12-19T05:46:05.273702Z","iopub.status.idle":"2024-12-19T05:46:05.289074Z","shell.execute_reply.started":"2024-12-19T05:46:05.273654Z","shell.execute_reply":"2024-12-19T05:46:05.288099Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Hypertuning","metadata":{}},{"cell_type":"code","source":"# Generalized objective function\ndef optuna_objective(trial, model_name, X_train, y_train, X_valid, y_valid):\n    if model_name == \"LightGBM\":\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1),\n            'max_depth': trial.suggest_int('max_depth', 6, 15),\n            'num_leaves': trial.suggest_int('num_leaves', 20, 500),\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 10, 50),\n            'feature_fraction': trial.suggest_float('feature_fraction', 0.7, 1.0),\n            'bagging_fraction': trial.suggest_float('bagging_fraction', 0.7, 1.0),\n            'bagging_freq': trial.suggest_int('bagging_freq', 1, 10),\n            'lambda_l1': trial.suggest_float('lambda_l1', 0.0, 10.0),\n            'lambda_l2': trial.suggest_float('lambda_l2', 0.0, 10.0),\n            'device': 'gpu'\n        }\n        model = LGBMRegressor(**params)\n    \n    elif model_name == \"XGBoost\":\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1),\n            'max_depth': trial.suggest_int('max_depth', 3, 12),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 300),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'reg_alpha': trial.suggest_float('reg_alpha', 0.0, 10.0),\n            'reg_lambda': trial.suggest_float('reg_lambda', 0.0, 10.0),\n            'random_state': 42,\n            'tree_method': 'gpu_hist'\n        }\n        model = XGBRegressor(**params)\n\n    elif model_name == \"CatBoost\":\n        params = {\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1),\n            'depth': trial.suggest_int('depth', 4, 10),\n            'iterations': trial.suggest_int('iterations', 100, 500),\n            'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1.0, 10.0),\n            'verbose': trial.suggest_int('verbose', 0, 2),\n            'task_type': 'GPU',\n            'random_seed': 42,\n        }\n        model = CatBoostRegressor(**params)\n\n    elif model_name == \"TabNet\":\n        params = {\n            'n_d': trial.suggest_int('n_d', 16, 128, step=16),\n            'n_a': trial.suggest_int('n_a', 16, 128, step=16),\n            'n_steps': trial.suggest_int('n_steps', 3, 10),\n            'gamma': trial.suggest_float('gamma', 1.0, 2.0, step=0.1),\n            'n_independent': trial.suggest_int('n_independent', 1, 5),\n            'n_shared': trial.suggest_int('n_independent', 1, 5),\n            'lambda_sparse': trial.suggest_float('lambda_sparse', 1e-5, 1e-3, log=True),\n            'optimizer_fn': torch.optim.Adam,\n            'optimizer_params': dict(lr=trial.suggest_float('lr', 1e-4, 1e-2, log=True), weight_decay=1e-5),\n            'mask_type': 'entmax',\n            'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n            'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n            'verbose': trial.suggest_int('verbose', 0, 2),\n            'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n        }\n        model = TabNetRegressor(**params)\n\n    else:\n        raise ValueError(\"Unsupported model_name\")\n\n    clear_output(wait=True)  # Clear output before model training\n    if model_name == \"LightGBM\":\n        model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], callbacks=[early_stopping(stopping_rounds=20)])\n    elif model_name == \"TabNet\":\n        X_train = X_train.values\n        X_valid = X_valid.values\n        y_train = y_train.values.reshape(-1, 1)\n        y_valid = y_valid.values.reshape(-1, 1)\n        model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], patience=20, max_epochs=200)\n    else:\n        model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], early_stopping_rounds=20, verbose=0)\n\n    preds = model.predict(X_valid)\n    metric = mean_squared_error(y_valid, preds, squared=False)  # Use RMSE as the evaluation metric\n    \n    return metric\n\n# Optuna optimization function\ndef optimize_model(model_name, X_train, y_train, X_valid, y_valid, n_trials=50):\n    study = optuna.create_study(direction=\"minimize\")\n    study.optimize(lambda trial: optuna_objective(trial, model_name, X_train, y_train, X_valid, y_valid), n_trials=n_trials)\n    \n    return study.best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.290242Z","iopub.execute_input":"2024-12-19T05:46:05.291170Z","iopub.status.idle":"2024-12-19T05:46:05.308351Z","shell.execute_reply.started":"2024-12-19T05:46:05.291130Z","shell.execute_reply":"2024-12-19T05:46:05.307435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Uncomment for hypertuned parameters\n# Split the dataset into training and testing sets\n# target_col = \"sii\"\n# features = [col for col in feats if col != target_col]\n# X_train, X_valid, y_train, y_valid = train_test_split(\n#     train[features], train[target_col], test_size=0.2, random_state=SEED\n# )\n\n# Params = optimize_model(\"LightGBM\", X_train, y_train, X_valid, y_valid)\n# XGB_Params = optimize_model(\"XGBoost\", X_train, y_train, X_valid, y_valid)\n# CatBoost_Params = optimize_model(\"CatBoost\", X_train, y_train, X_valid, y_valid)\n# TabNet_Params = optimize_model(\"TabNet\", X_train, y_train, X_valid, y_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.309348Z","iopub.execute_input":"2024-12-19T05:46:05.309634Z","iopub.status.idle":"2024-12-19T05:46:05.320092Z","shell.execute_reply.started":"2024-12-19T05:46:05.309608Z","shell.execute_reply":"2024-12-19T05:46:05.319368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check\n# TabNet_Params = {}\n# print(f\"LGB Params: {Params}\\n\\nXGB Params: {XGB_Params}\\n\\nCatBoost Params: {CatBoost_Params}\\n\\nTabNet Params: {TabNet_Params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.321120Z","iopub.execute_input":"2024-12-19T05:46:05.321412Z","iopub.status.idle":"2024-12-19T05:46:05.328879Z","shell.execute_reply.started":"2024-12-19T05:46:05.321380Z","shell.execute_reply":"2024-12-19T05:46:05.327993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.09091561699334874,\n    'max_depth': 7,\n    'num_leaves': 231,\n    'min_data_in_leaf': 43,\n    'feature_fraction': 0.7321811423868281,\n    'bagging_fraction': 0.7906404539496502,\n    'bagging_freq': 5,\n    'lambda_l1': 2.6246589676128163,\n    'lambda_l2': 2.213721575702823\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.025693439488707764,\n    'max_depth': 3,\n    'n_estimators': 247,\n    'subsample': 0.7821481379919165,\n    'colsample_bytree': 0.7907289374830526,\n    'reg_alpha': 3.7912823039798114,\n    'reg_lambda': 3.5754695830472825,\n    'random_state': SEED\n}\n\n# CatBoost parameters\nCatBoost_Params = {\n    'learning_rate': 0.07370515915369293,\n    'depth': 8,\n    'iterations': 408,\n    'random_seed': SEED,\n    'cat_features': cat_cols,\n    'verbose': 0,\n    'l2_leaf_reg': 4.85351342315636\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.329935Z","iopub.execute_input":"2024-12-19T05:46:05.330173Z","iopub.status.idle":"2024-12-19T05:46:05.340599Z","shell.execute_reply.started":"2024-12-19T05:46:05.330150Z","shell.execute_reply":"2024-12-19T05:46:05.339879Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 1","metadata":{}},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nsubmission_1 = TrainML(voting_model, train, test)\n\n# Check\nsubmission_1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:46:05.341618Z","iopub.execute_input":"2024-12-19T05:46:05.341957Z","iopub.status.idle":"2024-12-19T05:49:19.486504Z","shell.execute_reply.started":"2024-12-19T05:46:05.341921Z","shell.execute_reply":"2024-12-19T05:49:19.485651Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 2","metadata":{}},{"cell_type":"code","source":"# Create an imputer to handle missing values using the median strategy\nimputer = SimpleImputer(strategy='median')\n\n# Define an ensemble of models using VotingRegressor, where each model pipeline includes \n# an imputer for missing values and a regressor (LGBM, XGBoost, CatBoost, RandomForest, GradientBoosting)\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[\n        ('imputer', imputer),\n        ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[\n        ('imputer', imputer),\n        ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[\n        ('imputer', imputer), \n        ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[\n        ('imputer', imputer),\n        ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[\n        ('imputer', imputer), \n        ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\n# Apply the ensemble model to train and predict on the data\nsubmission_2 = TrainML(ensemble, train, test)\n\n# Check\nsubmission_2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:49:19.487492Z","iopub.execute_input":"2024-12-19T05:49:19.487746Z","iopub.status.idle":"2024-12-19T05:51:22.591735Z","shell.execute_reply.started":"2024-12-19T05:49:19.487721Z","shell.execute_reply":"2024-12-19T05:51:22.590868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encoding & Feature Engineering","metadata":{}},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    \"\"\"\n    Defines an autoencoder neural network with an encoder and decoder.\n    The encoder reduces input dimensions to a lower encoding dimension, \n    while the decoder reconstructs the input from the encoding.\n    \"\"\"\n    def __init__(self, input_dim, encoding_dim):\n        # Initialize encoder and decoder layers\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        # Forward pass: encode and decode the input\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        \n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    \"\"\"\n    Scales the input dataframe, trains the autoencoder on the data, \n    and returns the encoded representation of the input data.\n    \"\"\"\n    # Standardize the input data\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    # Define loss function and optimizer\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    # Training loop\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        # Print loss every 10 epochs\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    # Get the encoded data after training\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    # Convert encoded data into a DataFrame\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:51:22.593161Z","iopub.execute_input":"2024-12-19T05:51:22.593486Z","iopub.status.idle":"2024-12-19T05:51:22.603354Z","shell.execute_reply.started":"2024-12-19T05:51:22.593424Z","shell.execute_reply":"2024-12-19T05:51:22.602524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    \"\"\"\n    Perform feature engineering on a dataset, including imputation for missing values.\n\n    Parameters:\n        df (pd.DataFrame): The input dataset.\n\n    Returns:\n        pd.DataFrame: The transformed dataset with new features and imputation applied.\n    \"\"\"\n\n    # Drop season columns (equally distributed)\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n\n    # Incorporated features by Ichigo_E\n    # Reference: https://www.kaggle.com/code/ichigoe/lb0-494-with-tabnet/notebook\n    df['Age_BP'] = df['Basic_Demos-Age'] * df['Physical-Systolic_BP'] # Blood pressure trends with age\n    df['Sex_HR'] = df['Basic_Demos-Sex'] * df['Physical-HeartRate'] # Cardiovascular patterns by sex\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age'] # BMI scaled by age\n    df['BMI_PAQC'] = df['Physical-BMI'] * df['PAQ_C-PAQ_C_Total'] # Physical activity scaled by body composition\n    \n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday'] # Interaction between BMI and Internet Hours\n    df['Age_Internet_Hours'] =  df['Basic_Demos-Age'] * df['PreInt_EduHx-computerinternet_hoursday'] # Internet hours scaled by age\n    df['PAQC_Internet_Hours'] = df['PAQ_C-PAQ_C_Total'] * df['PreInt_EduHx-computerinternet_hoursday'] # Internet hours scaled by physical activity\n    df['Sleep_Internet_Hours'] = df['SDS-SDS_Total_T'] * df['PreInt_EduHx-computerinternet_hoursday'] # Internet hours scaled by sleep disturbance\n    df['PAQC_CGAS'] = df['PAQ_C-PAQ_C_Total'] * df['CGAS-CGAS_Score'] # Global assessment vs. physical activity\n    df['Sleep_CGAS'] = df['SDS-SDS_Total_T'] * df['CGAS-CGAS_Score'] # Global assessment vs. sleep disturbances\n    \n    df['BP_HR'] = df['Physical-Systolic_BP'] * df['Physical-HeartRate'] # Cardiovascular health metrics\n    df['BP_prof'] = df['Physical-Systolic_BP'] * df['Physical-Diastolic_BP'] # Overall blood pressure profile\n    df['BMR_HR'] = df['BIA-BIA_BMR'] / df['Physical-HeartRate'] # Basal Metabolic Rate scaled by heart rate\n    df['HR_Sleep'] = df['Physical-HeartRate'] / df['SDS-SDS_Total_T'] # Heart rate vs. sleep disturbances\n    \n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI'] # Body Fat Percentage and BMI ratio\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat'] # Fat-Free Mass Index and Body Fat Percentage ratio\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat'] # Fat Mass Index and Body Fat Percentage ratio\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW'] # Lean Soft Tissue to Total Body Water ratio\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR'] # Body Fat Percentage multiplied by Basal Metabolic Rate\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE'] # Body Fat Percentage multiplied by Daily Energy Expenditure\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight'] # Basal Metabolic Rate scaled by weight\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight'] # Daily Energy Expenditure scaled by weight\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height'] # Skeletal Muscle Mass scaled by height\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI'] # Muscle-to-Fat Ratio\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight'] # Hydration Status\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW'] # Intracellular Water to Total Body Water ratio\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:51:22.604722Z","iopub.execute_input":"2024-12-19T05:51:22.605226Z","iopub.status.idle":"2024-12-19T05:51:22.621162Z","shell.execute_reply.started":"2024-12-19T05:51:22.605188Z","shell.execute_reply":"2024-12-19T05:51:22.620357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare training and test data by dropping 'id' column\ndf_train = s_train.drop('id', axis=1)\ndf_test = s_test.drop('id', axis=1)\n\n# Apply autoencoder to encode the training and test data\ntrain_encoded = perform_autoencoder(df_train, encoding_dim=60, \n                                    epochs=100, batch_size=32)\ntest_encoded = perform_autoencoder(df_test, encoding_dim=60, \n                                   epochs=100, batch_size=32)\n\n# Add 'id' column back to encoded data for merging later\nencoded_cols = train_encoded.columns.tolist()\ntrain_encoded[\"id\"] = s_train[\"id\"]\ntest_encoded['id'] = s_test[\"id\"]\n\n# Merge encoded features with original data based on 'id'\ntrain = pd.merge(train_df, train_encoded, how=\"left\", on='id')\ntest = pd.merge(test_df, test_encoded, how=\"left\", on='id')\n\n# Impute missing numeric values using KNN imputation\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\n# Combine imputed numeric data with non-numeric columns\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\n# Assign imputed data back to the train DataFrame\ntrain = train_imputed\n\n# Perform feature engineering on train and test data\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)  # Drop rows with too many missing values\ntest = feature_engineering(test)\n\n# Drop 'id' columns for model training\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\n# Define features for model training\nfe_cols = ['Age_BP', 'Sex_HR', 'BMI_Age', 'BMI_PAQC', 'BMI_Internet_Hours', \n           'Age_Internet_Hours', 'PAQC_Internet_Hours', 'Sleep_Internet_Hours', \n           'PAQC_CGAS', 'Sleep_CGAS', 'BP_HR', 'BP_prof', 'BMR_HR', 'HR_Sleep', \n           'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', \n           'BMR_Weight', 'DEE_Weight', 'SMM_Height', 'Muscle_to_Fat', \n           'Hydration_Status', 'ICW_TBW']\nfeatures = feats + fe_cols\nfeatures = [f for f in features if f not in cat_cols]\nfeaturesCols = features + encoded_cols\n\n# Select the relevant columns for training and testing\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')  # Ensure 'sii' has no missing values\n\nfeaturesCols.remove('sii')\ntest = test[featuresCols]\n\n# Filter infinity values\nif np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:51:22.622200Z","iopub.execute_input":"2024-12-19T05:51:22.622502Z","iopub.status.idle":"2024-12-19T05:51:40.957382Z","shell.execute_reply.started":"2024-12-19T05:51:22.622467Z","shell.execute_reply":"2024-12-19T05:51:40.956359Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 3","metadata":{}},{"cell_type":"code","source":"# Model parameters for LightGBM (GPU)\nParams = {\n    'learning_rate': 0.09091561699334874,\n    'max_depth': 7,\n    'num_leaves': 231,\n    'min_data_in_leaf': 43,\n    'feature_fraction': 0.7321811423868281,\n    'bagging_fraction': 0.7906404539496502,\n    'bagging_freq': 5,\n    'lambda_l1': 2.6246589676128163,\n    'lambda_l2': 2.213721575702823,\n    'device': 'gpu'\n}\n\n\n# XGBoost parameters (GPU)\nXGB_Params = {\n    'learning_rate': 0.025693439488707764,\n    'max_depth': 3,\n    'n_estimators': 247,\n    'subsample': 0.7821481379919165,\n    'colsample_bytree': 0.7907289374830526,\n    'reg_alpha': 3.7912823039798114,\n    'reg_lambda': 3.5754695830472825,\n    'random_state': SEED,\n    'tree_method': 'gpu_hist'\n}\n\n# CatBoost parameters (GPU)\nCatBoost_Params = {\n    'learning_rate': 0.07370515915369293,\n    'depth': 8,\n    'iterations': 408,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 4.85351342315636,\n    'task_type': 'GPU'\n}\n\n# TabNet hyperparameters (GPU)\nTabNet_Params = {\n    'n_d': 16,\n    'n_a': 112,\n    'n_steps': 5,\n    'gamma': 1.5,\n    'n_independent': 2,\n    'n_shared': 2,\n    'lambda_sparse': 0.0008861104664367309,\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': {\n        'lr': 0.008860623276288872, \n        'weight_decay': 1e-05\n    },\n    'mask_type': 'entmax',\n    'scheduler_params': {\n        'mode': 'min',\n        'patience': 10,\n        'min_lr': 1e-05,\n        'factor': 0.5\n    },\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:51:40.958738Z","iopub.execute_input":"2024-12-19T05:51:40.959486Z","iopub.status.idle":"2024-12-19T05:51:40.967677Z","shell.execute_reply.started":"2024-12-19T05:51:40.959422Z","shell.execute_reply":"2024-12-19T05:51:40.966776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n],weights=[4.0,4.0,5.0,4.0])\n\n# Train the ensemble model\nsubmission_3 = TrainML(voting_model, train, test)\n\n# Check\nsubmission_3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T05:51:40.969098Z","iopub.execute_input":"2024-12-19T05:51:40.969483Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ensemble & Submission","metadata":{}},{"cell_type":"code","source":"# Sort each submission by 'id' and reset the index to ensure consistent ordering\nsubmission_1 = submission_1.sort_values(by='id').reset_index(drop=True)\nsubmission_2 = submission_2.sort_values(by='id').reset_index(drop=True)\nsubmission_3 = submission_3.sort_values(by='id').reset_index(drop=True)\n\n# Combine the sorted submissions into a new DataFrame with 'id' and corresponding 'sii' values from each submission\ncombined = pd.DataFrame({\n    'id': sample['id'],\n    'sii_1': submission_1['sii'],\n    'sii_2': submission_2['sii'],\n    'sii_3': submission_3['sii']\n})\n\n# Define a function for majority voting to determine the final 'sii' value for each row\ndef majority_vote(row):\n    return row.mode()[0]\n\n# Apply majority voting to the three 'sii' columns and store the result in 'final_sii'\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\n# Prepare the final submission DataFrame with 'id' and 'final_sii' columns, renaming 'final_sii' to 'sii'\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\n# Save the final submission DataFrame to a CSV file\nfinal_submission.to_csv('submission.csv', index=False)\n\n# Print a confirmation message that the majority voting process is complete\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display result\nfinal_submission","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}