{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-12-11T16:49:00.998809Z","iopub.execute_input":"2024-12-11T16:49:00.999128Z","iopub.status.idle":"2024-12-11T16:49:43.531823Z","shell.execute_reply.started":"2024-12-11T16:49:00.999075Z","shell.execute_reply":"2024-12-11T16:49:43.530610Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nimport seaborn as sns\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, mean_squared_error\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.decomposition import PCA\nfrom sklearn.datasets import make_classification\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nfrom pytorch_tabnet.callbacks import Callback\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-11T16:49:55.834799Z","iopub.execute_input":"2024-12-11T16:49:55.835145Z","iopub.status.idle":"2024-12-11T16:50:25.859540Z","shell.execute_reply.started":"2024-12-11T16:49:55.835111Z","shell.execute_reply":"2024-12-11T16:50:25.858783Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T16:50:30.878211Z","iopub.execute_input":"2024-12-11T16:50:30.878908Z","iopub.status.idle":"2024-12-11T16:50:30.890733Z","shell.execute_reply.started":"2024-12-11T16:50:30.878875Z","shell.execute_reply":"2024-12-11T16:50:30.890019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n    \ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:41:35.060081Z","iopub.execute_input":"2024-12-11T02:41:35.060556Z","iopub.status.idle":"2024-12-11T02:43:01.924277Z","shell.execute_reply.started":"2024-12-11T02:41:35.060505Z","shell.execute_reply":"2024-12-11T02:43:01.923279Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:44:02.970964Z","iopub.execute_input":"2024-12-11T02:44:02.972670Z","iopub.status.idle":"2024-12-11T02:44:03.102708Z","shell.execute_reply.started":"2024-12-11T02:44:02.972610Z","shell.execute_reply":"2024-12-11T02:44:03.101657Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['int32', 'int64', 'float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:44:10.172301Z","iopub.execute_input":"2024-12-11T02:44:10.172688Z","iopub.status.idle":"2024-12-11T02:44:17.183204Z","shell.execute_reply.started":"2024-12-11T02:44:10.172656Z","shell.execute_reply":"2024-12-11T02:44:17.182422Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.drop('id', axis=1)\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:44:20.176367Z","iopub.execute_input":"2024-12-11T02:44:20.176775Z","iopub.status.idle":"2024-12-11T02:44:20.298430Z","shell.execute_reply.started":"2024-12-11T02:44:20.176744Z","shell.execute_reply":"2024-12-11T02:44:20.297367Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.drop('id', axis=1)\ntest","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:44:27.960133Z","iopub.execute_input":"2024-12-11T02:44:27.961007Z","iopub.status.idle":"2024-12-11T02:44:28.106943Z","shell.execute_reply.started":"2024-12-11T02:44:27.960970Z","shell.execute_reply":"2024-12-11T02:44:28.106011Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:44:33.493110Z","iopub.execute_input":"2024-12-11T02:44:33.493421Z","iopub.status.idle":"2024-12-11T02:44:33.505729Z","shell.execute_reply.started":"2024-12-11T02:44:33.493396Z","shell.execute_reply":"2024-12-11T02:44:33.504949Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:44:39.240193Z","iopub.execute_input":"2024-12-11T02:44:39.241212Z","iopub.status.idle":"2024-12-11T02:44:39.334192Z","shell.execute_reply.started":"2024-12-11T02:44:39.241173Z","shell.execute_reply":"2024-12-11T02:44:39.333347Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:44:42.257620Z","iopub.execute_input":"2024-12-11T02:44:42.258301Z","iopub.status.idle":"2024-12-11T02:44:42.387022Z","shell.execute_reply.started":"2024-12-11T02:44:42.258269Z","shell.execute_reply":"2024-12-11T02:44:42.386150Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"execution":{"iopub.status.busy":"2024-12-11T02:44:45.180700Z","iopub.execute_input":"2024-12-11T02:44:45.181546Z","iopub.status.idle":"2024-12-11T02:44:45.189254Z","shell.execute_reply.started":"2024-12-11T02:44:45.181494Z","shell.execute_reply":"2024-12-11T02:44:45.188597Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Find the feature importance","metadata":{}},{"cell_type":"code","source":"# def quadratic_weighted_kappa(y_true, y_pred):\n#     return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# def threshold_Rounder(oof_non_rounded, thresholds):\n#     return np.where(oof_non_rounded < thresholds[0], 0,\n#                     np.where(oof_non_rounded < thresholds[1], 1,\n#                              np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n# def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n#     rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n#     return -quadratic_weighted_kappa(y_true, rounded_p)\n\n# def TrainML(model_class, test_data, light_model, xgb_model, cat_model):\n#     X = train.drop(['sii'], axis=1)\n#     y = train['sii']\n\n#     SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n#     train_S = []\n#     test_S = []\n    \n#     oof_non_rounded = np.zeros(len(y), dtype=float) \n#     oof_rounded = np.zeros(len(y), dtype=int) \n#     test_preds = np.zeros((len(test_data), n_splits))\n\n#     for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n#         X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n#         y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n#         model = clone(model_class)\n#         model.fit(X_train, y_train)\n\n#         light_model.fit(X_train, y_train)\n#         xgb_model.fit(X_train, y_train)\n#         cat_model.fit(X_train, y_train)\n\n#         y_train_pred = model.predict(X_train)\n#         y_val_pred = model.predict(X_val)\n\n#         oof_non_rounded[test_idx] = y_val_pred\n#         y_val_pred_rounded = y_val_pred.round(0).astype(int)\n#         oof_rounded[test_idx] = y_val_pred_rounded\n\n#         train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n#         val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n#         train_S.append(train_kappa)\n#         test_S.append(val_kappa)\n        \n#         test_preds[:, fold] = model.predict(test_data)\n        \n#         print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n#         clear_output(wait=True)\n\n#     print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n#     print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n#     KappaOPtimizer = minimize(evaluate_predictions,\n#                               x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n#                               method='Nelder-Mead')\n#     assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n#     oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n#     tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n#     print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n#     tpm = test_preds.mean(axis=1)\n#     tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n#     submission = pd.DataFrame({\n#         'id': sample['id'],\n#         'sii': tpTuned\n#     })\n\n#     return submission","metadata":{"execution":{"iopub.status.busy":"2024-12-10T16:44:14.631531Z","iopub.execute_input":"2024-12-10T16:44:14.631849Z","iopub.status.idle":"2024-12-10T16:44:14.640992Z","shell.execute_reply.started":"2024-12-10T16:44:14.631819Z","shell.execute_reply":"2024-12-10T16:44:14.640035Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        X_imputed = self.imputer.fit_transform(X)\n            \n        if hasattr(y, 'values'):\n            y = y.values\n\n        X_train, X_valid, y_train, y_valid = train_test_split(X_imputed, y, test_size=0.2, random_state=SEED)\n\n        history = self.model.fit(\n            X_train = X_train,\n            y_train = y_train.reshape(-1, 1),\n            eval_set = [(X_valid, y_valid.reshape(-1, 1))],\n            eval_name = ['valid'],\n            eval_metric = ['mse'],\n            max_epochs = 200,\n            patience = 20,\n            batch_size = 1024,\n            virtual_batch_size = 128,\n            num_workers = 0,\n            drop_last = False,\n            callbacks = [\n                TabNetPretrainedModelCheckpoint(\n                    filepath = self.best_model_path,\n                    monitor = 'valid_mse',\n                    mode = 'min',\n                    save_best_only = True,\n                    verbose = True\n                )\n            ]\n        )\n\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)\n\n        return self\n\n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n\n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = self.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n\n        return result\n\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', save_best_only=True, verbose=1):\n        super().__init__()\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        if (self.mode == 'min' and current < self.best) or (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)","metadata":{"execution":{"iopub.status.busy":"2024-12-10T16:44:14.642472Z","iopub.execute_input":"2024-12-10T16:44:14.643020Z","iopub.status.idle":"2024-12-10T16:44:14.663842Z","shell.execute_reply.started":"2024-12-10T16:44:14.642980Z","shell.execute_reply":"2024-12-10T16:44:14.662691Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'cpu',\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU',\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params,random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)\n#ODT_Model = ObliqueDecisionTreeRegressor(**ODT_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model),\n    #('odt', ODT_Model),\n], weights=[4.0,4.0,5.0,4.0])","metadata":{"execution":{"iopub.status.busy":"2024-12-10T16:44:14.665589Z","iopub.execute_input":"2024-12-10T16:44:14.665965Z","iopub.status.idle":"2024-12-10T16:44:14.688410Z","shell.execute_reply.started":"2024-12-10T16:44:14.665906Z","shell.execute_reply":"2024-12-10T16:44:14.687681Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# light_model = voting_model.estimators[0][1]\n# xgb_model = voting_model.estimators[1][1]\n# cat_model = voting_model.estimators[2][1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.689365Z","iopub.execute_input":"2024-12-10T16:44:14.689604Z","iopub.status.idle":"2024-12-10T16:44:14.697788Z","shell.execute_reply.started":"2024-12-10T16:44:14.689580Z","shell.execute_reply":"2024-12-10T16:44:14.696924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Submission1 = TrainML(voting_model, test, light_model, xgb_model, cat_model)\n\n# Save submission\n# Submission1.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-10T16:44:14.698749Z","iopub.execute_input":"2024-12-10T16:44:14.699038Z","iopub.status.idle":"2024-12-10T16:44:14.708298Z","shell.execute_reply.started":"2024-12-10T16:44:14.699014Z","shell.execute_reply":"2024-12-10T16:44:14.707604Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Submission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.709345Z","iopub.execute_input":"2024-12-10T16:44:14.709638Z","iopub.status.idle":"2024-12-10T16:44:14.718453Z","shell.execute_reply.started":"2024-12-10T16:44:14.709613Z","shell.execute_reply":"2024-12-10T16:44:14.717693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X = train.drop(['sii'], axis=1)\n# # Feature importance từ LightGBM\n# light_feature_importance = pd.DataFrame({\n#     'Feature': X.columns,\n#     'Importance': light_model.feature_importances_\n# }).sort_values(by='Importance', ascending=False)\n\n# # Feature importance từ XGBoost\n# xgb_feature_importance = pd.DataFrame({\n#     'Feature': X.columns,\n#     'Importance': xgb_model.feature_importances_\n# }).sort_values(by='Importance', ascending=False)\n\n# # Feature importance từ CatBoost\n# cat_feature_importance = pd.DataFrame({\n#     'Feature': X.columns,\n#     'Importance': cat_model.get_feature_importance()\n# }).sort_values(by='Importance', ascending=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.719443Z","iopub.execute_input":"2024-12-10T16:44:14.719721Z","iopub.status.idle":"2024-12-10T16:44:14.728964Z","shell.execute_reply.started":"2024-12-10T16:44:14.719694Z","shell.execute_reply":"2024-12-10T16:44:14.728440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Normalize giá trị của mỗi mô hình\n# light_feature_importance['Importance'] /= light_feature_importance['Importance'].sum()\n# cat_feature_importance['Importance'] /= cat_feature_importance['Importance'].sum()\n# xgb_feature_importance['Importance'] /= xgb_feature_importance['Importance'].sum()\n\n# Nếu muốn chuẩn hóa dựa trên giá trị lớn nhất\n# light_importance['Importance'] /= light_importance['Importance'].max()\n# cat_importance['Importance'] /= cat_importance['Importance'].max()\n# xgb_importance['Importance'] /= xgb_importance['Importance'].max()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.730153Z","iopub.execute_input":"2024-12-10T16:44:14.730777Z","iopub.status.idle":"2024-12-10T16:44:14.742121Z","shell.execute_reply.started":"2024-12-10T16:44:14.730738Z","shell.execute_reply":"2024-12-10T16:44:14.741485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plt.figure(figsize=(12, 30))\n# sns.barplot(\n#     x=light_feature_importance['Importance'],\n#     y=light_feature_importance['Feature'],\n#     palette='viridis'\n# )\n# plt.title(\"Feature Importances (LightGBM)\", fontsize=16)\n# plt.xlabel(\"Importance\", fontsize=14)\n# plt.ylabel(\"Feature\", fontsize=14)\n# plt.xticks(fontsize=12)\n# plt.yticks(fontsize=10)  # Giảm cỡ chữ của nhãn trục y để phù hợp với nhiều đặc trưng\n# plt.tight_layout()  # Đảm bảo rằng biểu đồ không bị cắt\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.743061Z","iopub.execute_input":"2024-12-10T16:44:14.743350Z","iopub.status.idle":"2024-12-10T16:44:14.752908Z","shell.execute_reply.started":"2024-12-10T16:44:14.743307Z","shell.execute_reply":"2024-12-10T16:44:14.751987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plt.figure(figsize=(12, 30))\n# sns.barplot(\n#     x=cat_feature_importance['Importance'],\n#     y=cat_feature_importance['Feature'],\n#     palette='viridis'\n# )\n# plt.title(\"Feature Importances (CatBoost)\", fontsize=16)\n# plt.xlabel(\"Importance\", fontsize=14)\n# plt.ylabel(\"Feature\", fontsize=14)\n# plt.xticks(fontsize=12)\n# plt.yticks(fontsize=10)  # Giảm cỡ chữ của nhãn trục y để phù hợp với nhiều đặc trưng\n# plt.tight_layout()  # Đảm bảo rằng biểu đồ không bị cắt\n# plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.753987Z","iopub.execute_input":"2024-12-10T16:44:14.754355Z","iopub.status.idle":"2024-12-10T16:44:14.764329Z","shell.execute_reply.started":"2024-12-10T16:44:14.754315Z","shell.execute_reply":"2024-12-10T16:44:14.763442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plt.figure(figsize=(12, 30))\n# sns.barplot(\n#     x=xgb_feature_importance['Importance'],\n#     y=xgb_feature_importance['Feature'],\n#     palette='viridis'\n# )\n# plt.title(\"Feature Importances (XgBoost)\", fontsize=16)\n# plt.xlabel(\"Importance\", fontsize=14)\n# plt.ylabel(\"Feature\", fontsize=14)\n# plt.xticks(fontsize=12)\n# plt.yticks(fontsize=10)  # Giảm cỡ chữ của nhãn trục y để phù hợp với nhiều đặc trưng\n# plt.tight_layout()  # Đảm bảo rằng biểu đồ không bị cắt\n# plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.765445Z","iopub.execute_input":"2024-12-10T16:44:14.765711Z","iopub.status.idle":"2024-12-10T16:44:14.776682Z","shell.execute_reply.started":"2024-12-10T16:44:14.765689Z","shell.execute_reply":"2024-12-10T16:44:14.775976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# combined_importance = pd.DataFrame({\n#     'LightGBM': light_feature_importance.set_index('Feature')['Importance'],\n#     'XGBoost': xgb_feature_importance.set_index('Feature')['Importance'],\n#     'CatBoost': cat_feature_importance.set_index('Feature')['Importance']\n# }).fillna(0)\n# combined_importance = combined_importance.reset_index()\n# combined_importance","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.777707Z","iopub.execute_input":"2024-12-10T16:44:14.778707Z","iopub.status.idle":"2024-12-10T16:44:14.786334Z","shell.execute_reply.started":"2024-12-10T16:44:14.778679Z","shell.execute_reply":"2024-12-10T16:44:14.785653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Tính trung bình importance từ cả 3 mô hình\n# combined_importance = pd.DataFrame({\n#     'LightGBM': light_feature_importance.set_index('Feature')['Importance'],\n#     'XGBoost': xgb_feature_importance.set_index('Feature')['Importance'],\n#     'CatBoost': cat_feature_importance.set_index('Feature')['Importance']\n# }).fillna(0)\n# combined_importance = combined_importance.reset_index()\n\n# combined_importance['Mean_Importance'] = combined_importance[['LightGBM', 'XGBoost', 'CatBoost']].mean(axis=1)\n# combined_importance = combined_importance.sort_values(by='Mean_Importance', ascending=False)\n\n# # Trực quan hóa\n# plt.figure(figsize=(12, 30))\n# sns.barplot(\n#     x=combined_importance['Mean_Importance'],\n#     y=combined_importance['Feature'],\n#     palette='viridis'\n# )\n# plt.title(\"Feature Importances\", fontsize=16)\n# plt.xlabel(\"Importance\", fontsize=14)\n# plt.ylabel(\"Feature\", fontsize=14)\n# plt.xticks(fontsize=12)\n# plt.yticks(fontsize=10)  # Giảm cỡ chữ của nhãn trục y để phù hợp với nhiều đặc trưng\n# plt.tight_layout()  # Đảm bảo rằng biểu đồ không bị cắt\n# plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.787280Z","iopub.execute_input":"2024-12-10T16:44:14.787525Z","iopub.status.idle":"2024-12-10T16:44:14.795675Z","shell.execute_reply.started":"2024-12-10T16:44:14.787501Z","shell.execute_reply":"2024-12-10T16:44:14.794742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Tính giá trị trung bình của cột 'Mean_Importance'\n# mean_value = combined_importance['Mean_Importance'].mean()\n\n# # Lọc ra các feature có 'Mean_Importance' >= mean_value\n# filtered_importance = combined_importance[combined_importance['Mean_Importance'] >= mean_value]\n\n# # Sắp xếp các feature theo 'Mean_Importance' giảm dần và chỉ lấy cột 'Feature'\n# sorted_features = filtered_importance.sort_values(by='Mean_Importance', ascending=False)['Feature']\n\n# # Đưa kết quả về dạng list\n# sorted_features_list = sorted_features.tolist()\n\n# # In ra kết quả\n# print(sorted_features_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.796708Z","iopub.execute_input":"2024-12-10T16:44:14.797241Z","iopub.status.idle":"2024-12-10T16:44:14.807969Z","shell.execute_reply.started":"2024-12-10T16:44:14.797217Z","shell.execute_reply":"2024-12-10T16:44:14.807188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# reversed_features = combined_importance['Feature'][::-1].tolist()\n# reversed_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.808885Z","iopub.execute_input":"2024-12-10T16:44:14.809187Z","iopub.status.idle":"2024-12-10T16:44:14.818197Z","shell.execute_reply.started":"2024-12-10T16:44:14.809165Z","shell.execute_reply":"2024-12-10T16:44:14.817123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# reserved_features = ['Enc_55',\n#  'Enc_3',\n#  'Enc_2',\n#  'Enc_19',\n#  'BIA-BIA_FFM',\n#  'Enc_11',\n#  'Enc_49',\n#  'Enc_44',\n#  'Enc_57',\n#  'Enc_59',\n#  'Enc_60',\n#  'Enc_22',\n#  'Enc_35',\n#  'Enc_50',\n#  'Enc_33',\n#  'Enc_1',\n#  'Enc_26',\n#  'Enc_24',\n#  'Enc_10',\n#  'Enc_27',\n#  'Enc_7',\n#  'Enc_54',\n#  'Enc_58',\n#  'Enc_4',\n#  'Enc_46',\n#  'Enc_43',\n#  'Enc_48',\n#  'Enc_38',\n#  'Enc_28',\n#  'Enc_8',\n#  'Enc_16',\n#  'Enc_34',\n#  'Enc_31',\n#  'Enc_21',\n#  'Enc_39',\n#  'Enc_42',\n#  'Enc_14',\n#  'Enc_13',\n#  'Enc_9',\n#  'Enc_32',\n#  'Enc_36',\n#  'Enc_12',\n#  'Enc_25',\n#  'Enc_51',\n#  'Enc_6',\n#  'Enc_30',\n#  'Enc_17',\n#  'Enc_20',\n#  'Enc_41',\n#  'Enc_56',\n#  'Enc_45',\n#  'Enc_47',\n#  'Enc_5',\n#  'Enc_29',\n#  'Enc_18',\n#  'BIA-BIA_BMR',\n#  'FGC-FGC_SRR_Zone',\n#  'FGC-FGC_CU_Zone',\n#  'Enc_52',\n#  'BFP_BMR',\n#  'BIA-BIA_FMI',\n#  'Enc_40',\n#  'BFP_BMI',\n#  'Enc_23',\n#  'Enc_15',\n#  'BIA-BIA_Fat',\n#  'Enc_53',\n#  'BIA-BIA_TBW',\n#  'BIA-BIA_BMI',\n#  'FGC-FGC_GSND_Zone',\n#  'BIA-BIA_SMM',\n#  'BIA-BIA_LST',\n#  'BIA-BIA_ECW',\n#  'BIA-BIA_Frame_num',\n#  'FGC-FGC_GSD_Zone',\n#  'Enc_37',\n#  'FFMI_BFP',\n#  'ICW_TBW',\n#  'FGC-FGC_SRL',\n#  'BIA-BIA_FFMI',\n#  'BIA-BIA_LDM',\n#  'BIA-BIA_Activity_Level_num',\n#  'FGC-FGC_PU',\n#  'BMR_Weight',\n#  'Physical-BMI',\n#  'Physical-Diastolic_BP',\n#  'Physical-Weight',\n#  'FGC-FGC_TL_Zone',\n#  'SMM_Height',\n#  'FGC-FGC_PU_Zone',\n#  'BIA-BIA_DEE',\n#  'BIA-BIA_BMC',\n#  'BFP_DEE',\n#  'FGC-FGC_SRL_Zone',\n#  'Physical-Waist_Circumference',\n#  'DEE_Weight',\n#  'Hydration_Status',\n#  'BMI_PHR',\n#  'Physical-HeartRate',\n#  'FGC-FGC_GSND',\n#  'Fitness_Endurance-Max_Stage',\n#  'Muscle_to_Fat',\n#  'FGC-FGC_SRR',\n#  'BMI_Internet_Hours',\n#  'BMI_Age',\n#  'FGC-FGC_TL',\n#  'LST_TBW',\n#  'CGAS-CGAS_Score',\n#  'FGC-FGC_CU',\n#  'Physical-Systolic_BP',\n#  'FGC-FGC_GSD',\n#  'Fitness_Endurance-Time_Mins',\n#  'PAQ_C-PAQ_C_Total',\n#  'Physical-Height',\n#  'Basic_Demos-Sex',\n#  'PreInt_EduHx-computerinternet_hoursday',\n#  'Fitness_Endurance-Time_Sec',\n#  'FMI_BFP',\n#  'PAQ_A-PAQ_A_Total',\n#  'Basic_Demos-Age',\n#  'SDS-SDS_Total_T',\n#  'SDS-SDS_Total_Raw',\n#  'BIA-BIA_ICW',\n#  'Internet_Hours_Age']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T12:52:04.589531Z","iopub.execute_input":"2024-12-11T12:52:04.590177Z","iopub.status.idle":"2024-12-11T12:52:04.608561Z","shell.execute_reply.started":"2024-12-11T12:52:04.590132Z","shell.execute_reply":"2024-12-11T12:52:04.607678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Chạy log thử với các feature importance từ cao đến thấp","metadata":{}},{"cell_type":"code","source":"# sorted_features = ['Internet_Hours_Age', 'BIA-BIA_ICW', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'Basic_Demos-Age', 'PAQ_A-PAQ_A_Total', 'FMI_BFP', 'Fitness_Endurance-Time_Sec', 'PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Sex', 'Physical-Height', 'PAQ_C-PAQ_C_Total', 'Fitness_Endurance-Time_Mins', 'FGC-FGC_GSD', 'Physical-Systolic_BP', 'FGC-FGC_CU', 'CGAS-CGAS_Score', 'LST_TBW', 'FGC-FGC_TL', 'BMI_Age', 'BMI_Internet_Hours', 'FGC-FGC_SRR', 'Muscle_to_Fat', 'Fitness_Endurance-Max_Stage', 'FGC-FGC_GSND', 'Physical-HeartRate', 'BMI_PHR', 'Hydration_Status', 'DEE_Weight', 'Physical-Waist_Circumference', 'FGC-FGC_SRL_Zone', 'BFP_DEE', 'BIA-BIA_BMC', 'BIA-BIA_DEE', 'FGC-FGC_PU_Zone', 'SMM_Height', 'FGC-FGC_TL_Zone', 'Physical-Weight', 'Physical-Diastolic_BP', 'Physical-BMI', 'BMR_Weight', 'FGC-FGC_PU', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_LDM', 'BIA-BIA_FFMI', 'FGC-FGC_SRL', 'ICW_TBW', 'FFMI_BFP']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.837195Z","iopub.execute_input":"2024-12-10T16:44:14.837961Z","iopub.status.idle":"2024-12-10T16:44:14.852065Z","shell.execute_reply.started":"2024-12-10T16:44:14.837905Z","shell.execute_reply":"2024-12-10T16:44:14.850988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def quadratic_weighted_kappa(y_true, y_pred):\n#     return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# def threshold_Rounder(oof_non_rounded, thresholds):\n#     return np.where(oof_non_rounded < thresholds[0], 0,\n#                     np.where(oof_non_rounded < thresholds[1], 1,\n#                              np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n# def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n#     rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n#     return -quadratic_weighted_kappa(y_true, rounded_p)\n\n# def TrainML_LogTest(model_class, test_data, first_score, selected_features, feature_tKappa_dict):\n#     for feature in sorted_features:\n#         print(f\"Processing feature: {feature}\")\n#         # Tạo bản sao của train\n#         train_transformed = train.copy()\n    \n#         # Áp dụng np.log1p() lên feature hiện tại\n#         train_transformed[feature] = np.log1p(train[feature])\n        \n#         X = train_transformed.drop(['sii'], axis=1)\n#         y = train_transformed['sii']\n    \n#         SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n        \n#         train_S = []\n#         test_S = []\n        \n#         oof_non_rounded = np.zeros(len(y), dtype=float) \n#         oof_rounded = np.zeros(len(y), dtype=int) \n#         test_preds = np.zeros((len(test_data), n_splits))\n    \n#         for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n#             X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n#             y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n    \n#             model = clone(model_class)\n#             model.fit(X_train, y_train)\n    \n#             y_train_pred = model.predict(X_train)\n#             y_val_pred = model.predict(X_val)\n    \n#             oof_non_rounded[test_idx] = y_val_pred\n#             y_val_pred_rounded = y_val_pred.round(0).astype(int)\n#             oof_rounded[test_idx] = y_val_pred_rounded\n    \n#             train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n#             val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n    \n#             train_S.append(train_kappa)\n#             test_S.append(val_kappa)\n            \n#             test_preds[:, fold] = model.predict(test_data)\n            \n#             print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n#             clear_output(wait=True)\n    \n#         print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n#         print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    \n#         KappaOPtimizer = minimize(evaluate_predictions,\n#                                   x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n#                                   method='Nelder-Mead')\n#         assert KappaOPtimizer.success, \"Optimization did not converge.\"\n        \n#         oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n#         tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    \n#         print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n    \n#         if(tKappa >= first_score):\n#             selected_features.append(feature)\n#             feature_tKappa_dict[feature] = tKappa\n#         else: continue","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.854052Z","iopub.execute_input":"2024-12-10T16:44:14.854561Z","iopub.status.idle":"2024-12-10T16:44:14.869685Z","shell.execute_reply.started":"2024-12-10T16:44:14.854512Z","shell.execute_reply":"2024-12-10T16:44:14.868761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# first_score = 0.519\n# selected_features = []\n# feature_tKappa_dict = {}\n# TrainML_LogTest(voting_model, test, first_score, selected_features, feature_tKappa_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.870504Z","iopub.execute_input":"2024-12-10T16:44:14.870757Z","iopub.status.idle":"2024-12-10T16:44:14.884566Z","shell.execute_reply.started":"2024-12-10T16:44:14.870734Z","shell.execute_reply":"2024-12-10T16:44:14.883202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# selected_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.885687Z","iopub.execute_input":"2024-12-10T16:44:14.886068Z","iopub.status.idle":"2024-12-10T16:44:14.894767Z","shell.execute_reply.started":"2024-12-10T16:44:14.886020Z","shell.execute_reply":"2024-12-10T16:44:14.893994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# feature_tKappa_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.895762Z","iopub.execute_input":"2024-12-10T16:44:14.896096Z","iopub.status.idle":"2024-12-10T16:44:14.902914Z","shell.execute_reply.started":"2024-12-10T16:44:14.896060Z","shell.execute_reply":"2024-12-10T16:44:14.902197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Sắp xếp các feature theo tKappa giảm dần\n# sorted_log_features = sorted(feature_tKappa_dict.items(), key=lambda item: item[1], reverse=True)\n# sorted_feature_names = [item[0] for item in sorted_log_features]\n\n# # In danh sách feature\n# print(sorted_feature_names)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.903803Z","iopub.execute_input":"2024-12-10T16:44:14.904139Z","iopub.status.idle":"2024-12-10T16:44:14.912892Z","shell.execute_reply.started":"2024-12-10T16:44:14.904111Z","shell.execute_reply":"2024-12-10T16:44:14.912178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sorted_feature_names = ['PAQ_C-PAQ_C_Total', 'LST_TBW', 'SMM_Height', 'FGC-FGC_PU', 'Physical-Weight', 'Physical-BMI', 'Physical-Waist_Circumference', 'FGC-FGC_SRR', 'Basic_Demos-Sex', 'FGC-FGC_GSD', 'CGAS-CGAS_Score', 'BMR_Weight', 'DEE_Weight', 'BIA-BIA_BMC', 'Physical-Systolic_BP', 'SDS-SDS_Total_Raw', 'Basic_Demos-Age', 'FGC-FGC_SRL_Zone', 'BIA-BIA_DEE', 'Muscle_to_Fat', 'BMI_Age', 'BFP_DEE', 'FGC-FGC_CU', 'Physical-HeartRate', 'FGC-FGC_PU_Zone', 'Physical-Diastolic_BP', 'BIA-BIA_Activity_Level_num', 'ICW_TBW', 'BMI_PHR', 'BIA-BIA_ICW', 'FGC-FGC_SRL', 'BMI_Internet_Hours', 'Fitness_Endurance-Time_Sec', 'FMI_BFP', 'SDS-SDS_Total_T', 'Fitness_Endurance-Time_Mins', 'Internet_Hours_Age', 'FGC-FGC_TL_Zone', 'FFMI_BFP', 'BIA-BIA_FFMI', 'FGC-FGC_TL', 'BIA-BIA_LDM', 'FGC-FGC_GSND', 'Hydration_Status', 'Fitness_Endurance-Max_Stage', 'PreInt_EduHx-computerinternet_hoursday', 'PAQ_A-PAQ_A_Total', 'Physical-Height']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T03:00:43.655544Z","iopub.execute_input":"2024-12-11T03:00:43.656069Z","iopub.status.idle":"2024-12-11T03:00:43.664201Z","shell.execute_reply.started":"2024-12-11T03:00:43.656013Z","shell.execute_reply":"2024-12-11T03:00:43.663189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def quadratic_weighted_kappa(y_true, y_pred):\n#     return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# def threshold_Rounder(oof_non_rounded, thresholds):\n#     return np.where(oof_non_rounded < thresholds[0], 0,\n#                     np.where(oof_non_rounded < thresholds[1], 1,\n#                              np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n# def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n#     rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n#     return -quadratic_weighted_kappa(y_true, rounded_p)\n\n# def TrainML_Log(model_class, test_data, score, log_features):\n#     for feature in sorted_feature_names:\n#         print(f\"Processing feature: {feature}\")\n#         train_transformed = train.copy()\n#         train_transformed[feature] = np.log1p(train[feature])\n    \n#         X = train_transformed.drop(['sii'], axis=1)\n#         y = train_transformed['sii']\n    \n#         SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n        \n#         train_S = []\n#         test_S = []\n        \n#         oof_non_rounded = np.zeros(len(y), dtype=float) \n#         oof_rounded = np.zeros(len(y), dtype=int) \n#         test_preds = np.zeros((len(test_data), n_splits))\n    \n#         for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n#             X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n#             y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n    \n#             model = clone(model_class)\n#             model.fit(X_train, y_train)\n    \n#             y_train_pred = model.predict(X_train)\n#             y_val_pred = model.predict(X_val)\n    \n#             oof_non_rounded[test_idx] = y_val_pred\n#             y_val_pred_rounded = y_val_pred.round(0).astype(int)\n#             oof_rounded[test_idx] = y_val_pred_rounded\n    \n#             train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n#             val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n    \n#             train_S.append(train_kappa)\n#             test_S.append(val_kappa)\n            \n#             test_preds[:, fold] = model.predict(test_data)\n            \n#             print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n#             clear_output(wait=True)\n    \n#         print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n#         print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    \n#         KappaOPtimizer = minimize(evaluate_predictions,\n#                                   x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n#                                   method='Nelder-Mead')\n#         assert KappaOPtimizer.success, \"Optimization did not converge.\"\n        \n#         oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n#         tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    \n#         print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n    \n#         if(tKappa >= score):\n#             score = tKappa\n#             train[feature] = np.log1p(train[feature])\n#             log_features.append(feature)\n#         else: continue","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.924165Z","iopub.execute_input":"2024-12-10T16:44:14.924540Z","iopub.status.idle":"2024-12-10T16:44:14.936646Z","shell.execute_reply.started":"2024-12-10T16:44:14.924500Z","shell.execute_reply":"2024-12-10T16:44:14.936059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# score = 0.519\n# log_features = []\n# TrainML_Log(voting_model, test, score, log_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.937478Z","iopub.execute_input":"2024-12-10T16:44:14.937718Z","iopub.status.idle":"2024-12-10T16:44:14.952207Z","shell.execute_reply.started":"2024-12-10T16:44:14.937693Z","shell.execute_reply":"2024-12-10T16:44:14.951376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\nX_vals1 = []\nY_vals1 = []\nY_preds1 = []\n\ndef TrainML(model_class, test_data):\n    # for feature in sorted_feature_names:\n    #     if feature in train.columns:\n    #         train[feature] = np.log1p(train[feature])\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        X_vals1.append(X_val)\n        Y_vals1.append(y_val)\n        Y_preds1.append(y_val_pred)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.953401Z","iopub.execute_input":"2024-12-10T16:44:14.953636Z","iopub.status.idle":"2024-12-10T16:44:14.966335Z","shell.execute_reply.started":"2024-12-10T16:44:14.953613Z","shell.execute_reply":"2024-12-10T16:44:14.965530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission1 = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:44:14.967235Z","iopub.execute_input":"2024-12-10T16:44:14.967495Z","iopub.status.idle":"2024-12-10T16:45:34.742600Z","shell.execute_reply.started":"2024-12-10T16:44:14.967472Z","shell.execute_reply":"2024-12-10T16:45:34.741472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission1","metadata":{"execution":{"iopub.status.busy":"2024-12-10T16:45:34.744289Z","iopub.execute_input":"2024-12-10T16:45:34.744723Z","iopub.status.idle":"2024-12-10T16:45:34.754873Z","shell.execute_reply.started":"2024-12-10T16:45:34.744676Z","shell.execute_reply":"2024-12-10T16:45:34.753919Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-12-11T16:50:53.791515Z","iopub.execute_input":"2024-12-11T16:50:53.792309Z","iopub.status.idle":"2024-12-11T16:52:08.258635Z","shell.execute_reply.started":"2024-12-11T16:50:53.792274Z","shell.execute_reply":"2024-12-11T16:52:08.257668Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n    \nX_vals2 = []\nY_vals2 = []\nY_preds2 = []\n\ndef TrainML(model_class, test_data):\n    # for feature in sorted_feature_names:\n    #     if feature in train.columns:\n    #         train[feature] = np.log1p(train[feature])\n    # train_transformed = train.copy()\n    # train_transformed[common_feature[0]] = np.log1p(train_transformed[common_feature[0]])\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        X_vals2.append(X_val)\n        Y_vals2.append(y_val)\n        Y_preds2.append(y_val_pred)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)\n#ODT_Model = ObliqueDecisionTreeRegressor(**ODT_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    #('tabnet', TabNet_Model),\n    #('odt', ODT_Model),\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T03:31:58.872147Z","iopub.execute_input":"2024-12-11T03:31:58.872801Z","iopub.status.idle":"2024-12-11T03:31:59.168952Z","shell.execute_reply.started":"2024-12-11T03:31:58.872767Z","shell.execute_reply":"2024-12-11T03:31:59.167844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:47:43.963514Z","iopub.execute_input":"2024-12-10T16:47:43.963898Z","iopub.status.idle":"2024-12-10T16:47:43.973701Z","shell.execute_reply.started":"2024-12-10T16:47:43.963855Z","shell.execute_reply":"2024-12-10T16:47:43.972668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\nX_vals3 = []\nY_vals3 = []\nY_preds3 = []\n\ndef TrainML(model_class, test_data):\n    # for feature in sorted_feature_names:\n    #     if feature in train.columns:\n    #         train[feature] = np.log1p(train[feature])\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        X_vals3.append(X_val)\n        Y_vals3.append(y_val)\n        Y_preds3.append(y_val_pred)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb',    Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb',    Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat',    Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf',     Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb',     Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))])),\n    #('tabnet', Pipeline(steps=[('imputer', imputer), ('regressor', TabNetWrapper(**TabNet_Params))])),\n    #('odt',    Pipeline(steps=[('imputer', imputer), ('regressor', ObliqueDecisionTreeRegressor(**ODT_Params))])),\n])\n\nSubmission3 = TrainML(ensemble, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:47:43.975340Z","iopub.execute_input":"2024-12-10T16:47:43.975761Z","iopub.status.idle":"2024-12-10T16:51:06.151418Z","shell.execute_reply.started":"2024-12-10T16:47:43.975715Z","shell.execute_reply":"2024-12-10T16:51:06.150491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:51:06.152862Z","iopub.execute_input":"2024-12-10T16:51:06.153241Z","iopub.status.idle":"2024-12-10T16:51:06.165800Z","shell.execute_reply.started":"2024-12-10T16:51:06.153201Z","shell.execute_reply":"2024-12-10T16:51:06.164776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:51:06.167136Z","iopub.execute_input":"2024-12-10T16:51:06.167688Z","iopub.status.idle":"2024-12-10T16:51:06.190882Z","shell.execute_reply.started":"2024-12-10T16:51:06.167651Z","shell.execute_reply":"2024-12-10T16:51:06.190014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T16:51:06.192062Z","iopub.execute_input":"2024-12-10T16:51:06.192813Z","iopub.status.idle":"2024-12-10T16:51:06.204019Z","shell.execute_reply.started":"2024-12-10T16:51:06.192769Z","shell.execute_reply":"2024-12-10T16:51:06.203188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}