{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import Dependencies","metadata":{}},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T20:35:41.132164Z","iopub.execute_input":"2024-12-15T20:35:41.132533Z","iopub.status.idle":"2024-12-15T20:36:21.918512Z","shell.execute_reply.started":"2024-12-15T20:35:41.132494Z","shell.execute_reply":"2024-12-15T20:36:21.917263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom concurrent.futures import ThreadPoolExecutor\nimport os\nimport torch\nfrom tqdm import tqdm\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, StackingRegressor\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import mean_squared_error, cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nfrom pytorch_tabnet.callbacks import Callback\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comm_dir = \"/kaggle/input/child-mind-institute-problematic-internet-use\"\nSEED = 42\nfeatures_cols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncategorical_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n                   'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n                   'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(f\"{comm_dir}/train.csv\")\ntest_df = pd.read_csv(f\"{comm_dir}/test.csv\")\nsubmission_df = pd.read_csv(f\"{comm_dir}/sample_submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.columns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"col_list = []\nfor col in train_df.columns:\n    if col == 'sii':\n        continue\n        \n    if col not in test_df.columns:\n        col_list.append(col)\ncol_list","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.drop(col_list, axis=1, inplace=True)\ntrain_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_counts = train_df.isna().sum()\nnan_counts1 = test_df.isna().sum()\nprint(nan_counts)\nprint(\"---------\")\nprint(nan_counts1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ts.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.merge(train_df, train_ts, how=\"left\", on=\"id\")\ntest_df = pd.merge(test_df, test_ts, how=\"left\", on=\"id\")\n\ntrain_df = train_df.drop([\"id\"], axis=1)\ntest_df = test_df.drop([\"id\"], axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_cols += time_series_cols\ntrain_df = train_df[features_cols]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.dropna(subset=\"sii\")\ntrain_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_df.loc[:, categorical_col] = train_df.loc[:, categorical_col].fillna('Missing')\n#train_df.loc[:, categorical_col] = train_df.loc[:, categorical_col].astype('category')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mappings = {'Spring':0, 'Summer':1, 'Fall':2, 'Winter':3}\n\nfor col in categorical_cols:\n    train_df[col] = train_df[col].map(mappings).fillna(-100) \n    train_df[col] = train_df[col].astype('int64')\n    test_df[col] = test_df[col].map(mappings).fillna(-100) \n    test_df[col] = test_df[col].astype('int64')\n\ntrain_df = train_df.reset_index(drop=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = train_df.drop(['sii'], axis=1)\ny_train = train_df['sii']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.dtypes)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        X_imputed = self.imputer.fit_transform(X)\n            \n        if hasattr(y, 'values'):\n            y = y.values\n\n        X_train, X_valid, y_train, y_valid = train_test_split(X_imputed, y, test_size=0.2, random_state=SEED)\n\n        history = self.model.fit(\n            X_train = X_train,\n            y_train = y_train.reshape(-1, 1),\n            eval_set = [(X_valid, y_valid.reshape(-1, 1))],\n            eval_name = ['valid'],\n            eval_metric = ['mse'],\n            max_epochs = 200,\n            patience = 20,\n            batch_size = 1024,\n            virtual_batch_size = 128,\n            num_workers = 0,\n            drop_last = False,\n            callbacks = [\n                TabNetPretrainedModelCheckpoint(\n                    filepath = self.best_model_path,\n                    monitor = 'valid_mse',\n                    mode = 'min',\n                    save_best_only = True,\n                    verbose = True\n                )\n            ]\n        )\n\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)\n\n        return self\n\n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n\n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = self.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n\n        return result\n\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', save_best_only=True, verbose=1):\n        super().__init__()\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        if (self.mode == 'min' and current < self.best) or (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y, y_pred):\n    rounded_pred = threshold_Rounder(y_pred, thresholds)\n    return -cohen_kappa_score(y, rounded_pred, weights=\"quadratic\")\n\ndef train(X, y, test_df):\n    kfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n    \n    lgbm = LGBMRegressor(              \n        learning_rate=0.04,       \n        n_estimators=250,          \n        num_leaves=420,          \n        min_data_in_leaf=15,      \n        max_depth=12,\n        random_state=SEED, \n        verbose=-1\n    )\n    \n    xgb = XGBRegressor(\n        learning_rate=0.05,\n        eta=0.5,\n        max_depth=7,\n        alpha=0.5,\n        num_parallel_tree=2,\n        n_estimators=250,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        reg_alpha=1, \n        reg_lambda=5,\n        random_state=SEED,\n        eval_metric='rmse')\n    \n    catboost = CatBoostRegressor(\n        iterations=750, \n        learning_rate=0.01, \n        depth=7, \n        l2_leaf_reg=7.31183636902306,\n        random_strength=1.7097065892440113, \n        bagging_temperature=0.026593521316435192,  \n        border_count=12,\n        random_state=SEED,\n        silent=True)\n    \n    tabnet = TabNetWrapper(**TabNet_Params)\n    \n    stacking_regressor = StackingRegressor(estimators=[\n    ('lgbm', lgbm),\n    ('xgb', xgb),\n    ('catboost', catboost),\n    ('tabnet', tabnet)\n    ])\n    \n    y_pred = np.zeros(len(y), dtype=float)\n    test_preds = np.zeros((len(test_df), 5))\n    \n    for fold, (train_idx, val_idx) in enumerate(tqdm(kfold.split(X, y), desc=\"Training Folds\", total=5)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        imputer = SimpleImputer(strategy='mean')\n        X_train = pd.DataFrame(imputer.fit_transform(X_train), columns=X_train.columns)\n        X_val = pd.DataFrame(imputer.transform(X_val), columns=X_val.columns)\n        \n        stacking_regressor.fit(X_train, y_train)\n        train_pred = stacking_regressor.predict(X_train)\n        val_pred = stacking_regressor.predict(X_val)\n        y_pred[val_idx] = val_pred\n        test_preds[:, fold] = stacking_regressor.predict(test_df)\n        fold_val_loss = cohen_kappa_score(y_val, val_pred.round(0).astype(int), weights=\"quadratic\")\n        fold_train_loss = cohen_kappa_score(y_train, train_pred.round(0).astype(int), weights=\"quadratic\")\n        print(f'Fold Train Loss: {fold_train_loss:.4f}')\n        print(f'Fold Val Loss: {fold_val_loss:.4f}')\n        \n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, y_pred), \n                              method='Nelder-Mead')\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': submission_df['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = train(X_train, y_train, test_df)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\nprint(submission)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}