{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:35.653517Z","iopub.execute_input":"2024-12-26T10:34:35.653741Z","iopub.status.idle":"2024-12-26T10:34:40.364745Z","shell.execute_reply.started":"2024-12-26T10:34:35.653713Z","shell.execute_reply":"2024-12-26T10:34:40.363680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:40.365738Z","iopub.execute_input":"2024-12-26T10:34:40.365992Z","iopub.status.idle":"2024-12-26T10:34:44.350278Z","shell.execute_reply.started":"2024-12-26T10:34:40.365963Z","shell.execute_reply":"2024-12-26T10:34:44.349388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import precision_score\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:44.351235Z","iopub.execute_input":"2024-12-26T10:34:44.351702Z","iopub.status.idle":"2024-12-26T10:34:56.368275Z","shell.execute_reply.started":"2024-12-26T10:34:44.351670Z","shell.execute_reply":"2024-12-26T10:34:56.367531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.370149Z","iopub.execute_input":"2024-12-26T10:34:56.370733Z","iopub.status.idle":"2024-12-26T10:34:56.379758Z","shell.execute_reply.started":"2024-12-26T10:34:56.370709Z","shell.execute_reply":"2024-12-26T10:34:56.378928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def process_file(filename, dirname):\n#     df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n#     df.drop('step', axis=1, inplace=True)\n#     return df.describe().values.reshape(-1), filename.split('=')[1]\n\n# def load_time_series(dirname) -> pd.DataFrame:\n#     ids = os.listdir(dirname)\n    \n#     with ThreadPoolExecutor() as executor:\n#         results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n#     stats, indexes = zip(*results)\n    \n#     df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n#     df['id'] = indexes\n#     return df\n\n\n# class AutoEncoder(nn.Module):\n#     def __init__(self, input_dim, encoding_dim):\n#         super(AutoEncoder, self).__init__()\n#         self.encoder = nn.Sequential(\n#             nn.Linear(input_dim, encoding_dim*3),\n#             nn.ReLU(),\n#             nn.Linear(encoding_dim*3, encoding_dim*2),\n#             nn.ReLU(),\n#             nn.Linear(encoding_dim*2, encoding_dim),\n#             nn.ReLU()\n#         )\n#         self.decoder = nn.Sequential(\n#             nn.Linear(encoding_dim, input_dim*2),\n#             nn.ReLU(),\n#             nn.Linear(input_dim*2, input_dim*3),\n#             nn.ReLU(),\n#             nn.Linear(input_dim*3, input_dim),\n#             nn.Sigmoid()\n#         )\n        \n#     def forward(self, x):\n#         encoded = self.encoder(x)\n#         decoded = self.decoder(encoded)\n#         return decoded\n\n\n# def perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n#     scaler = StandardScaler()\n#     df_scaled = scaler.fit_transform(df)\n    \n#     data_tensor = torch.FloatTensor(df_scaled)\n    \n#     input_dim = data_tensor.shape[1]\n#     autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n#     criterion = nn.MSELoss()\n#     optimizer = optim.Adam(autoencoder.parameters())\n    \n#     for epoch in range(epochs):\n#         for i in range(0, len(data_tensor), batch_size):\n#             batch = data_tensor[i : i + batch_size]\n#             optimizer.zero_grad()\n#             reconstructed = autoencoder(batch)\n#             loss = criterion(reconstructed, batch)\n#             loss.backward()\n#             optimizer.step()\n            \n#         if (epoch + 1) % 10 == 0:\n#             print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n#     with torch.no_grad():\n#         encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n#     df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n#     return df_encoded\n\n# def feature_engineering(df):\n#     season_cols = [col for col in df.columns if 'Season' in col]\n#     df = df.drop(season_cols, axis=1) \n#     df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n#     df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n#     df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n#     df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n#     df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n#     df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n#     df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n#     df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n#     df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n#     df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n#     df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n#     df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n#     df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n#     df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n#     df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n#     df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n#     return df\n\n# train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n# test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n# sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n# test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# df_train = train_ts.drop('id', axis=1)\n# df_test = test_ts.drop('id', axis=1)\n\n# train_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\n# test_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\n# time_series_cols = train_ts_encoded.columns.tolist()\n# train_ts_encoded[\"id\"]=train_ts[\"id\"]\n# test_ts_encoded['id']=test_ts[\"id\"]\n\n# train = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\n# test = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.381217Z","iopub.execute_input":"2024-12-26T10:34:56.381514Z","iopub.status.idle":"2024-12-26T10:34:56.396054Z","shell.execute_reply.started":"2024-12-26T10:34:56.381484Z","shell.execute_reply":"2024-12-26T10:34:56.395328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pca_columns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\npca_columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.396754Z","iopub.execute_input":"2024-12-26T10:34:56.397047Z","iopub.status.idle":"2024-12-26T10:34:56.418316Z","shell.execute_reply.started":"2024-12-26T10:34:56.397015Z","shell.execute_reply":"2024-12-26T10:34:56.417670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# imputer = KNNImputer(n_neighbors=5)\n# numeric_cols = train.select_dtypes(include=['int32', 'int64', 'float64', 'int64']).columns\n# imputed_data = imputer.fit_transform(train[numeric_cols])\n# train_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n# train_imputed[pca_columns] = train_imputed[pca_columns].round().astype(int)\n# for col in train.columns:\n#     if col not in numeric_cols:\n#         train_imputed[col] = train[col]\n        \n# train = train_imputed\n\n# train = feature_engineering(train)\n# train = train.dropna(thresh=10, axis=0)\n# test = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.419152Z","iopub.execute_input":"2024-12-26T10:34:56.419435Z","iopub.status.idle":"2024-12-26T10:34:56.433685Z","shell.execute_reply.started":"2024-12-26T10:34:56.419406Z","shell.execute_reply":"2024-12-26T10:34:56.432876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ID = train['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.434590Z","iopub.execute_input":"2024-12-26T10:34:56.434875Z","iopub.status.idle":"2024-12-26T10:34:56.450048Z","shell.execute_reply.started":"2024-12-26T10:34:56.434856Z","shell.execute_reply":"2024-12-26T10:34:56.449240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# first_train = train\n# first_test = test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.450853Z","iopub.execute_input":"2024-12-26T10:34:56.451140Z","iopub.status.idle":"2024-12-26T10:34:56.465619Z","shell.execute_reply.started":"2024-12-26T10:34:56.451105Z","shell.execute_reply":"2024-12-26T10:34:56.464777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for col in pca_columns:\n#     print(first_train[col].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.466503Z","iopub.execute_input":"2024-12-26T10:34:56.466788Z","iopub.status.idle":"2024-12-26T10:34:56.479658Z","shell.execute_reply.started":"2024-12-26T10:34:56.466759Z","shell.execute_reply":"2024-12-26T10:34:56.479016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission1 = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.480380Z","iopub.execute_input":"2024-12-26T10:34:56.480646Z","iopub.status.idle":"2024-12-26T10:34:56.495935Z","shell.execute_reply.started":"2024-12-26T10:34:56.480620Z","shell.execute_reply":"2024-12-26T10:34:56.495398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for col in tqdm(pca_columns, total=n_splits):\n#     featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n#                     'CGAS-CGAS_Score', 'Physical-BMI',\n#                     'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n#                     'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n#                     'Fitness_Endurance-Max_Stage',\n#                     'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n#                     'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n#                     'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n#                     'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n#                     'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n#                     'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n#                     'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n#                     'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n#                     'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n#                     'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n#                     'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n#                     'SDS-SDS_Total_T',\n#                     'PreInt_EduHx-computerinternet_hoursday', col, 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n#                     'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n#                     'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW','BMI_PHR']\n    \n#     featuresCols += time_series_cols\n    \n#     train = first_train[featuresCols]\n#     train = train.dropna(subset=col)\n    \n#     featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n#                     'CGAS-CGAS_Score', 'Physical-BMI',\n#                     'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n#                     'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n#                     'Fitness_Endurance-Max_Stage',\n#                     'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n#                     'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n#                     'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n#                     'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n#                     'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n#                     'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n#                     'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n#                     'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n#                     'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n#                     'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n#                     'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n#                     'SDS-SDS_Total_T',\n#                     'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n#                     'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n#                     'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW','BMI_PHR']\n    \n#     featuresCols += time_series_cols\n#     test = first_test[featuresCols]\n    \n#     if np.any(np.isinf(train)):\n#         train = train.replace([np.inf, -np.inf], np.nan)\n    \n#     def quadratic_weighted_kappa(y_true, y_pred):\n#         return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    \n#     def threshold_Rounder(oof_non_rounded, thresholds):\n#          return np.where(oof_non_rounded < thresholds[0], 0,\n#                     np.where(oof_non_rounded < thresholds[1], 1,\n#                              np.where(oof_non_rounded < thresholds[2], 2,\n#                                       np.where(oof_non_rounded < thresholds[3], 3,\n#                                                np.where(oof_non_rounded < thresholds[4], 4, 5)))))\n    \n#     def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n#         rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n#         return -quadratic_weighted_kappa(y_true, rounded_p)\n    \n#     def TrainML(model_class, test_data):\n#         X = train.drop([col], axis=1)\n#         y = train[col]\n        \n#         SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n#         train_S = []\n#         test_S = []\n        \n#         oof_non_rounded = np.zeros(len(y), dtype=float) \n#         oof_rounded = np.zeros(len(y), dtype=int) \n#         test_preds = np.zeros((len(test_data), n_splits))\n    \n#         for fold, (train_idx, test_idx) in enumerate(SKF.split(X, y)):\n#             X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n#             y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n#             model = clone(model_class)\n#             model.fit(X_train, y_train)\n        \n#             y_train_pred = model.predict(X_train)\n#             y_val_pred = model.predict(X_val)\n        \n#             oof_non_rounded[test_idx] = y_val_pred\n#             y_val_pred_rounded = y_val_pred.round(0).astype(int)\n#             oof_rounded[test_idx] = y_val_pred_rounded\n\n#             train_kappa = quadratic_weighted_kappa(y_train.astype(int), y_train_pred.round(0).astype(int))\n#             val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        \n#             train_S.append(train_kappa)\n#             test_S.append(val_kappa)\n        \n#             test_preds[:, fold] = model.predict(test_data)\n        \n#             print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n#             clear_output(wait=True)\n            \n#         print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n#         print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    \n#         KappaOPtimizer = minimize(evaluate_predictions,\n#                                   x0=[0.5, 1.5, 2.5, 3.5, 4.5], args=(y, oof_non_rounded), \n#                                   method='Nelder-Mead')\n#         assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n#         oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n#         tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    \n#         print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n    \n#         tpm = test_preds.mean(axis=1)\n#         tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n#         return tpTuned\n        \n#     # Model parameters for LightGBM\n#     Params = {\n#         'learning_rate': 0.046,\n#         'max_depth': 12,\n#         'num_leaves': 478,\n#         'min_data_in_leaf': 13,\n#         'feature_fraction': 0.893,\n#         'bagging_fraction': 0.784,\n#         'bagging_freq': 4,\n#         'lambda_l1': 10,  # Increased from 6.59\n#         'lambda_l2': 0.01,  # Increased from 2.68e-06\n#         'device': 'cpu'\n    \n#     }\n    \n    \n#     # XGBoost parameters\n#     XGB_Params = {\n#         'learning_rate': 0.05,\n#         'max_depth': 6,\n#         'n_estimators': 200,\n#         'subsample': 0.8,\n#         'colsample_bytree': 0.8,\n#         'reg_alpha': 1,  # Increased from 0.1\n#         'reg_lambda': 5,  # Increased from 1\n#         'random_state': SEED,\n#         'tree_method': 'gpu_hist',\n    \n#     }\n    \n    \n#     CatBoost_Params = {\n#         'learning_rate': 0.05,\n#         'depth': 6,\n#         'iterations': 200,\n#         'random_seed': SEED,\n#         'verbose': 0,\n#         'l2_leaf_reg': 10,  # Increase this value\n#         'task_type': 'GPU'\n    \n#     }\n#     class TabNetWrapper(BaseEstimator, RegressorMixin):\n#         def __init__(self, **kwargs):\n#             self.model = TabNetRegressor(**kwargs)\n#             self.kwargs = kwargs\n#             self.imputer = SimpleImputer(strategy='median')\n#             self.best_model_path = 'best_tabnet_model.pt'\n            \n#         def fit(self, X, y):\n#             # Handle missing values\n#             X_imputed = self.imputer.fit_transform(X)\n            \n#             if hasattr(y, 'values'):\n#                 y = y.values\n                \n#             # Create internal validation set\n#             X_train, X_valid, y_train, y_valid = train_test_split(\n#                 X_imputed, \n#                 y, \n#                 test_size=0.2,\n#                 random_state=42\n#             )\n            \n#             # Train TabNet model\n#             history = self.model.fit(\n#                 X_train=X_train,\n#                 y_train=y_train.reshape(-1, 1),\n#                 eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n#                 eval_name=['valid'],\n#                 eval_metric=['mse'],\n#                 max_epochs=200,\n#                 patience=20,\n#                 batch_size=1024,\n#                 virtual_batch_size=128,\n#                 num_workers=0,\n#                 drop_last=False,\n#                 callbacks=[\n#                     TabNetPretrainedModelCheckpoint(\n#                         filepath=self.best_model_path,\n#                         monitor='valid_mse',\n#                         mode='min',\n#                         save_best_only=True,\n#                         verbose=True\n#                     )\n#                 ]\n#             )\n            \n#             # Load the best model\n#             if os.path.exists(self.best_model_path):\n#                 self.model.load_model(self.best_model_path)\n#                 os.remove(self.best_model_path)  # Remove temporary file\n            \n#             return self\n        \n#         def predict(self, X):\n#             X_imputed = self.imputer.transform(X)\n#             return self.model.predict(X_imputed).flatten()\n        \n#         def __deepcopy__(self, memo):\n#             # Add deepcopy support for scikit-learn\n#             cls = self.__class__\n#             result = cls.__new__(cls)\n#             memo[id(self)] = result\n#             for k, v in self.__dict__.items():\n#                 setattr(result, k, deepcopy(v, memo))\n#             return result\n    \n#     # TabNet hyperparameters\n#     TabNet_Params = {\n#         'n_d': 64,              # Width of the decision prediction layer\n#         'n_a': 64,              # Width of the attention embedding for each step\n#         'n_steps': 5,           # Number of steps in the architecture\n#         'gamma': 1.5,           # Coefficient for feature selection regularization\n#         'n_independent': 2,     # Number of independent GLU layer in each GLU block\n#         'n_shared': 2,          # Number of shared GLU layer in each GLU block\n#         'lambda_sparse': 1e-4,  # Sparsity regularization\n#         'optimizer_fn': torch.optim.Adam,\n#         'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n#         'mask_type': 'entmax',\n#         'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n#         'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n#         'verbose': 1,\n#         'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n#     }\n    \n#     class TabNetPretrainedModelCheckpoint(Callback):\n#         def __init__(self, filepath, monitor='val_loss', mode='min', \n#                      save_best_only=True, verbose=1):\n#             super().__init__()  # Initialize parent class\n#             self.filepath = filepath\n#             self.monitor = monitor\n#             self.mode = mode\n#             self.save_best_only = save_best_only\n#             self.verbose = verbose\n#             self.best = float('inf') if mode == 'min' else -float('inf')\n            \n#         def on_train_begin(self, logs=None):\n#             self.model = self.trainer  # Use trainer itself as model\n            \n#         def on_epoch_end(self, epoch, logs=None):\n#             logs = logs or {}\n#             current = logs.get(self.monitor)\n#             if current is None:\n#                 return\n            \n#             # Check if current metric is better than best\n#             if (self.mode == 'min' and current < self.best) or \\\n#                (self.mode == 'max' and current > self.best):\n#                 if self.verbose:\n#                     print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n#                 self.best = current\n#                 if self.save_best_only:\n#                     self.model.save_model(self.filepath)  # Save the entire model\n    \n#     # Create model instances\n#     Light = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\n#     XGB_Model = XGBRegressor(**XGB_Params)\n#     CatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n#     TabNet_Model = TabNetWrapper(**TabNet_Params) # New\n    \n#     # Combine models using Voting Regressor\n#     voting_model = VotingRegressor(estimators=[\n#         ('lightgbm', Light),\n#         ('xgboost', XGB_Model),\n#         ('catboost', CatBoost_Model),\n#         ('tabnet', TabNet_Model),\n#         #('odt', ODT_Model),\n#     ], weights=[4.0,4.0,5.0,4.0])\n#     submission1[col] = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.496729Z","iopub.execute_input":"2024-12-26T10:34:56.496975Z","iopub.status.idle":"2024-12-26T10:34:56.513640Z","shell.execute_reply.started":"2024-12-26T10:34:56.496936Z","shell.execute_reply":"2024-12-26T10:34:56.512847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check = pd.DataFrame(submission1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.516623Z","iopub.execute_input":"2024-12-26T10:34:56.516851Z","iopub.status.idle":"2024-12-26T10:34:56.534494Z","shell.execute_reply.started":"2024-12-26T10:34:56.516831Z","shell.execute_reply":"2024-12-26T10:34:56.533798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission1 = pd.DataFrame({\n#     'id': sample['id'],\n#     'sii': check.sum(axis=1)\n# })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.536143Z","iopub.execute_input":"2024-12-26T10:34:56.536407Z","iopub.status.idle":"2024-12-26T10:34:56.549855Z","shell.execute_reply.started":"2024-12-26T10:34:56.536388Z","shell.execute_reply":"2024-12-26T10:34:56.549159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# s0 = submission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.550464Z","iopub.execute_input":"2024-12-26T10:34:56.550662Z","iopub.status.idle":"2024-12-26T10:34:56.563617Z","shell.execute_reply.started":"2024-12-26T10:34:56.550644Z","shell.execute_reply":"2024-12-26T10:34:56.563027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.564293Z","iopub.execute_input":"2024-12-26T10:34:56.564576Z","iopub.status.idle":"2024-12-26T10:34:56.577931Z","shell.execute_reply.started":"2024-12-26T10:34:56.564551Z","shell.execute_reply":"2024-12-26T10:34:56.577311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def classify_score(score):\n    if 0 <= score <= 30:\n        return 0\n    elif 31 <= score <= 49:\n        return 1\n    elif 50 <= score <= 79:\n        return 2\n    elif 80 <= score <= 100:\n        return 3\n    else:\n        return 4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.578596Z","iopub.execute_input":"2024-12-26T10:34:56.578777Z","iopub.status.idle":"2024-12-26T10:34:56.594929Z","shell.execute_reply.started":"2024-12-26T10:34:56.578762Z","shell.execute_reply":"2024-12-26T10:34:56.594204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission1['sii'] = submission1['sii'].map(classify_score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.595571Z","iopub.execute_input":"2024-12-26T10:34:56.595793Z","iopub.status.idle":"2024-12-26T10:34:56.610113Z","shell.execute_reply.started":"2024-12-26T10:34:56.595774Z","shell.execute_reply":"2024-12-26T10:34:56.609412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.610882Z","iopub.execute_input":"2024-12-26T10:34:56.611145Z","iopub.status.idle":"2024-12-26T10:34:56.626928Z","shell.execute_reply.started":"2024-12-26T10:34:56.611115Z","shell.execute_reply":"2024-12-26T10:34:56.626137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n# test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n# sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# def process_file(filename, dirname):\n#     df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n#     df.drop('step', axis=1, inplace=True)\n#     return df.describe().values.reshape(-1), filename.split('=')[1]\n\n# def load_time_series(dirname) -> pd.DataFrame:\n#     ids = os.listdir(dirname)\n    \n#     with ThreadPoolExecutor() as executor:\n#         results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n#     stats, indexes = zip(*results)\n    \n#     df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n#     df['id'] = indexes\n#     return df\n        \n# train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n# test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# time_series_cols = train_ts.columns.tolist()\n# time_series_cols.remove(\"id\")\n\n# train = pd.merge(train, train_ts, how=\"left\", on='id')\n# test = pd.merge(test, test_ts, how=\"left\", on='id')\n\n# train = train.drop('id', axis=1)\n# test = test.drop('id', axis=1)\n\n# first_train = train\n# first_test = test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.627765Z","iopub.execute_input":"2024-12-26T10:34:56.628046Z","iopub.status.idle":"2024-12-26T10:34:56.642936Z","shell.execute_reply.started":"2024-12-26T10:34:56.628020Z","shell.execute_reply":"2024-12-26T10:34:56.642294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission2 = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.643584Z","iopub.execute_input":"2024-12-26T10:34:56.643822Z","iopub.status.idle":"2024-12-26T10:34:56.661704Z","shell.execute_reply.started":"2024-12-26T10:34:56.643795Z","shell.execute_reply":"2024-12-26T10:34:56.660973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for Pca_col in tqdm(pca_columns, total=n_splits):\n#     featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n#                     'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n#                     'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n#                     'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n#                     'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n#                     'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n#                     'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n#                     'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n#                     'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n#                     'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n#                     'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n#                     'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n#                     'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n#                     'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n#                     'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n#                     'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n#                     'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n#                     'PreInt_EduHx-computerinternet_hoursday', Pca_col]\n    \n#     featuresCols += time_series_cols\n    \n#     train = first_train[featuresCols]\n#     train = train.dropna(subset=Pca_col)\n    \n#     cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n#               'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n#               'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n    \n#     def update(df):\n#         global cat_c\n#         for c in cat_c: \n#             df[c] = df[c].fillna('Missing')\n#             df[c] = df[c].astype('category')\n#         return df\n            \n#     train = update(train)\n#     test = update(test)\n    \n#     def create_mapping(column, dataset):\n#         unique_values = dataset[column].unique()\n#         return {value: idx for idx, value in enumerate(unique_values)}\n    \n#     for col in cat_c:\n#         mapping = create_mapping(col, train)\n#         mappingTe = create_mapping(col, test)\n        \n#         train[col] = train[col].replace(mapping).astype(int)\n#         test[col] = test[col].replace(mappingTe).astype(int)\n\n    \n#     def quadratic_weighted_kappa(y_true, y_pred):\n#         return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    \n#     def threshold_Rounder(oof_non_rounded, thresholds):\n#          return np.where(oof_non_rounded < thresholds[0], 0,\n#                     np.where(oof_non_rounded < thresholds[1], 1,\n#                              np.where(oof_non_rounded < thresholds[2], 2,\n#                                       np.where(oof_non_rounded < thresholds[3], 3,\n#                                                np.where(oof_non_rounded < thresholds[4], 4, 5)))))\n    \n#     def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n#         rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n#         return -quadratic_weighted_kappa(y_true, rounded_p)\n    \n#     def TrainML(model_class, test_data):\n        \n#         X = train.drop([Pca_col], axis=1)\n#         y = train[Pca_col]\n        \n#         SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n#         train_S = []\n#         test_S = []\n        \n#         oof_non_rounded = np.zeros(len(y), dtype=float) \n#         oof_rounded = np.zeros(len(y), dtype=int) \n#         test_preds = np.zeros((len(test_data), n_splits))\n    \n#         for fold, (train_idx, test_idx) in enumerate(SKF.split(X, y)):\n#             X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n#             y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n#             model = clone(model_class)\n#             model.fit(X_train, y_train)\n        \n#             y_train_pred = model.predict(X_train)\n#             y_val_pred = model.predict(X_val)\n        \n#             oof_non_rounded[test_idx] = y_val_pred\n#             y_val_pred_rounded = y_val_pred.round(0).astype(int)\n#             oof_rounded[test_idx] = y_val_pred_rounded\n\n#             train_kappa = quadratic_weighted_kappa(y_train.astype(int), y_train_pred.round(0).astype(int))\n#             val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        \n#             train_S.append(train_kappa)\n#             test_S.append(val_kappa)\n        \n#             test_preds[:, fold] = model.predict(test_data)\n        \n#             print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n#             clear_output(wait=True)\n#         print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n#         print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    \n#         KappaOPtimizer = minimize(evaluate_predictions,\n#                                   x0=[0.5, 1.5, 2.5, 3.5, 4.5], args=(y, oof_non_rounded), \n#                                   method='Nelder-Mead')\n#         assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n#         oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n#         tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    \n#         print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n    \n#         tpm = test_preds.mean(axis=1)\n#         tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n#         return tpTuned\n        \n#     # Model parameters for LightGBM\n#     Params = {\n#         'learning_rate': 0.046,\n#         'max_depth': 12,\n#         'num_leaves': 478,\n#         'min_data_in_leaf': 13,\n#         'feature_fraction': 0.893,\n#         'bagging_fraction': 0.784,\n#         'bagging_freq': 4,\n#         'lambda_l1': 10,  # Increased from 6.59\n#         'lambda_l2': 0.01  # Increased from 2.68e-06\n#     }\n    \n    \n#     # XGBoost parameters\n#     XGB_Params = {\n#         'learning_rate': 0.05,\n#         'max_depth': 6,\n#         'n_estimators': 200,\n#         'subsample': 0.8,\n#         'colsample_bytree': 0.8,\n#         'reg_alpha': 1,  # Increased from 0.1\n#         'reg_lambda': 5,  # Increased from 1\n#         'random_state': SEED\n#     }\n    \n    \n#     CatBoost_Params = {\n#         'learning_rate': 0.05,\n#         'depth': 6,\n#         'iterations': 200,\n#         'random_seed': SEED,\n#         'cat_features': cat_c,\n#         'verbose': 0,\n#         'l2_leaf_reg': 10  # Increase this value\n#     }\n    \n#     # Create model instances\n#     Light = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\n#     XGB_Model = XGBRegressor(**XGB_Params)\n#     CatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n    \n#     # Combine models using Voting Regressor\n#     voting_model = VotingRegressor(estimators=[\n#         ('lightgbm', Light),\n#         ('xgboost', XGB_Model),\n#         ('catboost', CatBoost_Model)\n#     ])\n    \n#     # Train the ensemble model\n#     submission2[Pca_col] = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.662380Z","iopub.execute_input":"2024-12-26T10:34:56.662581Z","iopub.status.idle":"2024-12-26T10:34:56.679377Z","shell.execute_reply.started":"2024-12-26T10:34:56.662563Z","shell.execute_reply":"2024-12-26T10:34:56.678608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check2 = pd.DataFrame(submission2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.680053Z","iopub.execute_input":"2024-12-26T10:34:56.680261Z","iopub.status.idle":"2024-12-26T10:34:56.698966Z","shell.execute_reply.started":"2024-12-26T10:34:56.680244Z","shell.execute_reply":"2024-12-26T10:34:56.698337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# s1 = submission2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.699617Z","iopub.execute_input":"2024-12-26T10:34:56.699819Z","iopub.status.idle":"2024-12-26T10:34:56.715485Z","shell.execute_reply.started":"2024-12-26T10:34:56.699803Z","shell.execute_reply":"2024-12-26T10:34:56.714678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission2 = pd.DataFrame({\n#     'id': sample['id'],\n#     'sii': check2.sum(axis=1)\n# })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.716305Z","iopub.execute_input":"2024-12-26T10:34:56.716506Z","iopub.status.idle":"2024-12-26T10:34:56.730006Z","shell.execute_reply.started":"2024-12-26T10:34:56.716486Z","shell.execute_reply":"2024-12-26T10:34:56.729147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission2['sii'] = submission2['sii'].map(classify_score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.730832Z","iopub.execute_input":"2024-12-26T10:34:56.731144Z","iopub.status.idle":"2024-12-26T10:34:56.746388Z","shell.execute_reply.started":"2024-12-26T10:34:56.731115Z","shell.execute_reply":"2024-12-26T10:34:56.745732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.747186Z","iopub.execute_input":"2024-12-26T10:34:56.747404Z","iopub.status.idle":"2024-12-26T10:34:56.763727Z","shell.execute_reply.started":"2024-12-26T10:34:56.747386Z","shell.execute_reply":"2024-12-26T10:34:56.763027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.764532Z","iopub.execute_input":"2024-12-26T10:34:56.764826Z","iopub.status.idle":"2024-12-26T10:34:56.780592Z","shell.execute_reply.started":"2024-12-26T10:34:56.764799Z","shell.execute_reply":"2024-12-26T10:34:56.779900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n    \ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n    \ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:34:56.781469Z","iopub.execute_input":"2024-12-26T10:34:56.781762Z","iopub.status.idle":"2024-12-26T10:36:07.477024Z","shell.execute_reply.started":"2024-12-26T10:34:56.781734Z","shell.execute_reply":"2024-12-26T10:36:07.475930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission3 = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:36:07.477870Z","iopub.execute_input":"2024-12-26T10:36:07.478217Z","iopub.status.idle":"2024-12-26T10:36:07.482767Z","shell.execute_reply.started":"2024-12-26T10:36:07.478186Z","shell.execute_reply":"2024-12-26T10:36:07.481702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"first_train = train\nfirst_test = test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:36:07.483756Z","iopub.execute_input":"2024-12-26T10:36:07.484144Z","iopub.status.idle":"2024-12-26T10:36:07.503947Z","shell.execute_reply.started":"2024-12-26T10:36:07.484117Z","shell.execute_reply":"2024-12-26T10:36:07.503227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for Pca_col in tqdm(pca_columns, total=n_splits):\n    featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                    'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                    'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                    'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                    'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                    'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                    'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                    'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                    'PreInt_EduHx-computerinternet_hoursday', Pca_col]\n    \n    cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n              'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n              'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n    \n    \n    featuresCols += time_series_cols\n    \n    train = first_train[featuresCols]\n    train = train.dropna(subset=Pca_col)\n    \n    def update(df):\n        global cat_c\n        for c in cat_c: \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n        return df\n    \n    train = update(train)\n    test = update(test)\n    \n    def create_mapping(column, dataset):\n        unique_values = dataset[column].unique()\n        return {value: idx for idx, value in enumerate(unique_values)}\n    \n    for col in cat_c:\n        mapping = create_mapping(col, train)\n        mappingTe = create_mapping(col, test)\n        \n        train[col] = train[col].replace(mapping).astype(int)\n        test[col] = test[col].replace(mappingTe).astype(int)\n\n    def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    \n    def threshold_Rounder(oof_non_rounded, thresholds):\n         return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2,\n                                      np.where(oof_non_rounded < thresholds[3], 3,\n                                               np.where(oof_non_rounded < thresholds[4], 4, 5)))))\n    \n    def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n        rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_p)\n    \n    def TrainML(model_class, test_data):\n        X = train.drop([Pca_col], axis=1)\n        y = train[Pca_col]\n    \n        SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n        train_S = []\n        test_S = []\n        \n        oof_non_rounded = np.zeros(len(y), dtype=float) \n        oof_rounded = np.zeros(len(y), dtype=int) \n        test_preds = np.zeros((len(test_data), n_splits))\n    \n        for fold, (train_idx, test_idx) in enumerate(SKF.split(X, y)):\n            X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n            y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n            model = clone(model_class)\n            model.fit(X_train, y_train)\n        \n            y_train_pred = model.predict(X_train)\n            y_val_pred = model.predict(X_val)\n        \n            oof_non_rounded[test_idx] = y_val_pred\n            y_val_pred_rounded = y_val_pred.round(0).astype(int)\n            oof_rounded[test_idx] = y_val_pred_rounded\n\n            train_kappa = quadratic_weighted_kappa(y_train.astype(int), y_train_pred.round(0).astype(int))\n            val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        \n            train_S.append(train_kappa)\n            test_S.append(val_kappa)\n        \n            test_preds[:, fold] = model.predict(test_data)\n        \n            print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n            clear_output(wait=True)\n        print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n        print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    \n        KappaOPtimizer = minimize(evaluate_predictions,\n                                  x0=[0.5, 1.5, 2.5, 3.5, 4.5], args=(y, oof_non_rounded), \n                                  method='Nelder-Mead')\n        assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n        tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    \n        print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n    \n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n        return tpTuned\n\n    imputer = SimpleImputer(strategy='median')\n    \n    ensemble = VotingRegressor(estimators=[\n        ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n        ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n        ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n        ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n        ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n    ])\n    \n    submission3[Pca_col] = TrainML(ensemble, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T10:36:07.505065Z","iopub.execute_input":"2024-12-26T10:36:07.505442Z","iopub.status.idle":"2024-12-26T11:16:21.914390Z","shell.execute_reply.started":"2024-12-26T10:36:07.505414Z","shell.execute_reply":"2024-12-26T11:16:21.913682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"check3 = pd.DataFrame(submission3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T11:16:21.915201Z","iopub.execute_input":"2024-12-26T11:16:21.915419Z","iopub.status.idle":"2024-12-26T11:16:21.919586Z","shell.execute_reply.started":"2024-12-26T11:16:21.915400Z","shell.execute_reply":"2024-12-26T11:16:21.918770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s2 = submission3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T11:16:21.920484Z","iopub.execute_input":"2024-12-26T11:16:21.920704Z","iopub.status.idle":"2024-12-26T11:16:21.938244Z","shell.execute_reply.started":"2024-12-26T11:16:21.920686Z","shell.execute_reply":"2024-12-26T11:16:21.937605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': check3.sum(axis=1)\n})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T11:16:21.939032Z","iopub.execute_input":"2024-12-26T11:16:21.939227Z","iopub.status.idle":"2024-12-26T11:16:21.955703Z","shell.execute_reply.started":"2024-12-26T11:16:21.939210Z","shell.execute_reply":"2024-12-26T11:16:21.954899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission3['sii'] = submission3['sii'].map(classify_score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T11:16:21.956406Z","iopub.execute_input":"2024-12-26T11:16:21.956585Z","iopub.status.idle":"2024-12-26T11:16:21.971979Z","shell.execute_reply.started":"2024-12-26T11:16:21.956570Z","shell.execute_reply":"2024-12-26T11:16:21.971148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission3.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T11:16:21.972874Z","iopub.execute_input":"2024-12-26T11:16:21.973186Z","iopub.status.idle":"2024-12-26T11:16:21.992304Z","shell.execute_reply.started":"2024-12-26T11:16:21.973159Z","shell.execute_reply":"2024-12-26T11:16:21.991483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sub1 = submission1\n# sub2 = submission2\n# sub3 = submission3\n\n# sub1 = sub1.sort_values(by='id').reset_index(drop=True)\n# sub2 = sub2.sort_values(by='id').reset_index(drop=True)\n# sub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\n# combined = pd.DataFrame({\n#     'id': sub1['id'],\n#     'sii_1': sub1['sii'],\n#     'sii_2': sub2['sii'],\n#     'sii_3': sub3['sii']\n# })\n\n# def majority_vote(row):\n#     # if row['sii_1'] != row['sii_2'] and row['sii_1'] != row['sii_3']:\n#     #     return int(row.median()) \n#     # else:\n#     return row.mode()[0]\n\n# combined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\n# final_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\n# final_submission.to_csv('submission.csv', index=False)\n\n# print(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T11:16:21.993055Z","iopub.execute_input":"2024-12-26T11:16:21.993243Z","iopub.status.idle":"2024-12-26T11:16:21.998542Z","shell.execute_reply.started":"2024-12-26T11:16:21.993227Z","shell.execute_reply":"2024-12-26T11:16:21.997744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# final_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T11:16:21.999414Z","iopub.execute_input":"2024-12-26T11:16:21.999648Z","iopub.status.idle":"2024-12-26T11:16:22.018277Z","shell.execute_reply.started":"2024-12-26T11:16:21.999629Z","shell.execute_reply":"2024-12-26T11:16:22.017642Z"}},"outputs":[],"execution_count":null}]}