{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nn_splits = 5\nSEED = 42","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:48:16.208427Z","iopub.execute_input":"2024-12-12T09:48:16.209358Z","iopub.status.idle":"2024-12-12T09:48:16.216286Z","shell.execute_reply.started":"2024-12-12T09:48:16.209318Z","shell.execute_reply":"2024-12-12T09:48:16.215219Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Loading","metadata":{}},{"cell_type":"markdown","source":"## Load tabular dataset","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:48:16.218269Z","iopub.execute_input":"2024-12-12T09:48:16.218985Z","iopub.status.idle":"2024-12-12T09:48:16.267148Z","shell.execute_reply.started":"2024-12-12T09:48:16.218931Z","shell.execute_reply":"2024-12-12T09:48:16.266425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:48:16.268215Z","iopub.execute_input":"2024-12-12T09:48:16.268576Z","iopub.status.idle":"2024-12-12T09:48:16.440364Z","shell.execute_reply.started":"2024-12-12T09:48:16.268539Z","shell.execute_reply":"2024-12-12T09:48:16.439539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['id'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:48:16.442216Z","iopub.execute_input":"2024-12-12T09:48:16.442630Z","iopub.status.idle":"2024-12-12T09:48:16.450293Z","shell.execute_reply.started":"2024-12-12T09:48:16.442584Z","shell.execute_reply":"2024-12-12T09:48:16.449249Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load timeseries data","metadata":{}},{"cell_type":"markdown","source":"1. **process_file**: This function process file timeseries, extract general information in the file like count, mean, std, min, 25%, 50%, 75% and max of each features and then the features matrix is flattened to a vector to represent the data in the file\n2. **load_time_series** Format and load all timeseries files after processed in a folder.","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:48:16.452867Z","iopub.execute_input":"2024-12-12T09:48:16.453194Z","iopub.status.idle":"2024-12-12T09:48:16.460600Z","shell.execute_reply.started":"2024-12-12T09:48:16.453161Z","shell.execute_reply":"2024-12-12T09:48:16.459526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:48:16.461917Z","iopub.execute_input":"2024-12-12T09:48:16.462604Z","iopub.status.idle":"2024-12-12T09:49:31.724926Z","shell.execute_reply.started":"2024-12-12T09:48:16.462565Z","shell.execute_reply":"2024-12-12T09:49:31.723844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:31.726179Z","iopub.execute_input":"2024-12-12T09:49:31.726462Z","iopub.status.idle":"2024-12-12T09:49:31.799522Z","shell.execute_reply.started":"2024-12-12T09:49:31.726433Z","shell.execute_reply":"2024-12-12T09:49:31.798620Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"After that, we encode time series data using an autoencoder, This autoencoder encode the timeseries general features to a compact vector that contain most important information about the file. \n\n+ **AutoEncoder**:define AutoEncoder mode. \n\n+ **perform_autoencoder**: train the AutoEncoder model and then return the encoded featuresures\nures","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom sklearn.preprocessing import StandardScaler\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.LeakyReLU(0.2)\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:31.800592Z","iopub.execute_input":"2024-12-12T09:49:31.800864Z","iopub.status.idle":"2024-12-12T09:49:31.807351Z","shell.execute_reply.started":"2024-12-12T09:49:31.800839Z","shell.execute_reply":"2024-12-12T09:49:31.806534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n\n    data_tensor = torch.FloatTensor(df_scaled)\n\n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n\n    criterion = F.smooth_l1_loss\n    optimizer = optim.Adam(autoencoder.parameters())\n\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n\n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:31.808674Z","iopub.execute_input":"2024-12-12T09:49:31.809037Z","iopub.status.idle":"2024-12-12T09:49:31.818748Z","shell.execute_reply.started":"2024-12-12T09:49:31.808990Z","shell.execute_reply":"2024-12-12T09:49:31.817817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded[\"id\"] = test_ts[\"id\"]\ntrain_ts = train_ts_encoded\ntest_ts = test_ts_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:31.819715Z","iopub.execute_input":"2024-12-12T09:49:31.820037Z","iopub.status.idle":"2024-12-12T09:49:40.607620Z","shell.execute_reply.started":"2024-12-12T09:49:31.820000Z","shell.execute_reply":"2024-12-12T09:49:40.606656Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Time series data is then merged to the tabular data","metadata":{}},{"cell_type":"code","source":"train = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.608856Z","iopub.execute_input":"2024-12-12T09:49:40.609149Z","iopub.status.idle":"2024-12-12T09:49:40.624186Z","shell.execute_reply.started":"2024-12-12T09:49:40.609120Z","shell.execute_reply":"2024-12-12T09:49:40.623113Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Drop id because we dont take it as a feature.","metadata":{}},{"cell_type":"code","source":"train = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.625761Z","iopub.execute_input":"2024-12-12T09:49:40.626212Z","iopub.status.idle":"2024-12-12T09:49:40.633906Z","shell.execute_reply.started":"2024-12-12T09:49:40.626171Z","shell.execute_reply":"2024-12-12T09:49:40.632953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.635122Z","iopub.execute_input":"2024-12-12T09:49:40.635507Z","iopub.status.idle":"2024-12-12T09:49:40.732351Z","shell.execute_reply.started":"2024-12-12T09:49:40.635451Z","shell.execute_reply":"2024-12-12T09:49:40.731425Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Filtering","metadata":{}},{"cell_type":"markdown","source":"Select the columns which is present in test data to train","metadata":{}},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.735701Z","iopub.execute_input":"2024-12-12T09:49:40.736019Z","iopub.status.idle":"2024-12-12T09:49:40.744376Z","shell.execute_reply.started":"2024-12-12T09:49:40.735987Z","shell.execute_reply":"2024-12-12T09:49:40.743357Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Drop the NaN sii value","metadata":{}},{"cell_type":"code","source":"train = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.745611Z","iopub.execute_input":"2024-12-12T09:49:40.745960Z","iopub.status.idle":"2024-12-12T09:49:40.756584Z","shell.execute_reply.started":"2024-12-12T09:49:40.745926Z","shell.execute_reply":"2024-12-12T09:49:40.755596Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Fill the missing categorical data with \"Missing\" ","metadata":{}},{"cell_type":"code","source":"cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.757630Z","iopub.execute_input":"2024-12-12T09:49:40.757913Z","iopub.status.idle":"2024-12-12T09:49:40.788476Z","shell.execute_reply.started":"2024-12-12T09:49:40.757883Z","shell.execute_reply":"2024-12-12T09:49:40.787771Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Create a mapping from string to integer to push data to the model (Use one hot encode instead)","metadata":{}},{"cell_type":"code","source":"def create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.789662Z","iopub.execute_input":"2024-12-12T09:49:40.790574Z","iopub.status.idle":"2024-12-12T09:49:40.834890Z","shell.execute_reply.started":"2024-12-12T09:49:40.790528Z","shell.execute_reply":"2024-12-12T09:49:40.833907Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Some features are combined to create new features, these features may make model learn better representation of the data","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n\n    drop_cols = ['FGC-FGC_GSND', 'FGC-FGC_SRL', 'Fitness_Endurance-Max_Stage', 'Physical-Waist_Circumference', 'Physical-BMI', 'BIA-BIA_BMI']\n    df = df.drop(drop_cols, axis=1)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.835799Z","iopub.execute_input":"2024-12-12T09:49:40.836042Z","iopub.status.idle":"2024-12-12T09:49:40.843152Z","shell.execute_reply.started":"2024-12-12T09:49:40.836017Z","shell.execute_reply":"2024-12-12T09:49:40.842190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = feature_engineering(train)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.844231Z","iopub.execute_input":"2024-12-12T09:49:40.844518Z","iopub.status.idle":"2024-12-12T09:49:40.867490Z","shell.execute_reply.started":"2024-12-12T09:49:40.844458Z","shell.execute_reply":"2024-12-12T09:49:40.866611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.868872Z","iopub.execute_input":"2024-12-12T09:49:40.869171Z","iopub.status.idle":"2024-12-12T09:49:40.881890Z","shell.execute_reply.started":"2024-12-12T09:49:40.869144Z","shell.execute_reply":"2024-12-12T09:49:40.880801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.883319Z","iopub.execute_input":"2024-12-12T09:49:40.884033Z","iopub.status.idle":"2024-12-12T09:49:40.960778Z","shell.execute_reply.started":"2024-12-12T09:49:40.883994Z","shell.execute_reply":"2024-12-12T09:49:40.959934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sii'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.961966Z","iopub.execute_input":"2024-12-12T09:49:40.962307Z","iopub.status.idle":"2024-12-12T09:49:40.969848Z","shell.execute_reply.started":"2024-12-12T09:49:40.962268Z","shell.execute_reply":"2024-12-12T09:49:40.968881Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training Function","metadata":{}},{"cell_type":"markdown","source":"**quadratic_weighted_kappa**: calculate QWK value","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.970686Z","iopub.execute_input":"2024-12-12T09:49:40.971017Z","iopub.status.idle":"2024-12-12T09:49:40.976966Z","shell.execute_reply.started":"2024-12-12T09:49:40.970973Z","shell.execute_reply":"2024-12-12T09:49:40.975836Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**threshold_Rounder**: Turn the sii from PCIAT_Total to categorical ","metadata":{}},{"cell_type":"code","source":"def threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.978065Z","iopub.execute_input":"2024-12-12T09:49:40.978403Z","iopub.status.idle":"2024-12-12T09:49:40.987381Z","shell.execute_reply.started":"2024-12-12T09:49:40.978359Z","shell.execute_reply":"2024-12-12T09:49:40.986522Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**evaluate_predictions**: this function evaluate the prediction of the model by first turn integer prediction values to categorical values and then calculate QWK from it and the true labels.","metadata":{}},{"cell_type":"code","source":"def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:40.988363Z","iopub.execute_input":"2024-12-12T09:49:40.988780Z","iopub.status.idle":"2024-12-12T09:49:40.997352Z","shell.execute_reply.started":"2024-12-12T09:49:40.988754Z","shell.execute_reply":"2024-12-12T09:49:40.996542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**TrainML**: Train the model using K-Fold, The model is regression model, predict a real value represent how bad the patient was. The value may not explicitly different, so we re-define the threshold to make it split more accurate","metadata":{}},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    # Apply K-Fold\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        # Train model\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        # Round to integer values\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        #Predict with test dataset\n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Using optimizer to find the best threshold\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    # Use the threshold retrive from the optimizer to predict again to evaluate\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    # Use the threshold retrive from the optimizer to predict test\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    # Create submition\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission, model, KappaOPtimizer.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:56:45.308043Z","iopub.execute_input":"2024-12-12T09:56:45.309021Z","iopub.status.idle":"2024-12-12T09:56:45.319860Z","shell.execute_reply.started":"2024-12-12T09:56:45.308974Z","shell.execute_reply":"2024-12-12T09:56:45.318938Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create model and train the model","metadata":{}},{"cell_type":"markdown","source":"**TabNet Model**: A transformer-based model that can auto preprocessing data by choosing most important features to retain and drop necessary features. Because of that, this model can be intepreter easily to visualize the important features, create an adventage for error analysis. To be used in this code, the model must be in the right format of sklearn. There are some class to do it:\n+ **SimpleImputer**: This class is to impute missing values.\n+ **BaseEstimator**: Inherit this class turn the child class to an estimator in sklearn, more information can be found [here](https://scikit-learn.org/1.5/modules/generated/sklearn.base.BaseEstimator.html)\n+ **RegressorMixin**: This class is an mixin class, which add some useful function about regression to the child class, more information can be found [here](https://scikit-learn.org/1.5/modules/generated/sklearn.base.RegressorMixin.html)\n+ **TabNetRegressor**: This is the raw tabnet model. For more information about config and model, please refer this [github repo](https://github.com/dreamquark-ai/tabnet) \n  \n**TabNetWrapper**: This class is sklearn representation of tabnet\n\n**TabNetPretrainedModelCheckpoint**: This class purpose is to save the tabnet model","metadata":{}},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:49:41.012293Z","iopub.execute_input":"2024-12-12T09:49:41.012571Z","iopub.status.idle":"2024-12-12T09:50:21.833281Z","shell.execute_reply.started":"2024-12-12T09:49:41.012546Z","shell.execute_reply":"2024-12-12T09:50:21.831922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = KNNImputer(n_neighbors=5)\n        #self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n\n    def fit(self, X, y):\n        X_imputed = self.imputer.fit_transform(X)\n\n        if hasattr(y, 'values'):\n            y = y.values\n\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed,\n            y,\n            test_size=0.2,\n            random_state=42\n        )\n\n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse', 'mae', 'rmse'],\n            max_epochs=500,\n            patience=50,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n\n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n\n        return self\n\n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n\n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n\nTabNet_Params = {\n    'n_d': 64,\n    'n_a': 64,\n    'n_steps': 5,\n    'gamma': 1.5,\n    'n_independent': 2,\n    'n_shared': 2,\n    'lambda_sparse': 1e-4,\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min',\n                 save_best_only=True, verbose=1):\n        super().__init__()\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n\n    def on_train_begin(self, logs=None):\n        self.model = self.trainer\n\n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:50:21.835514Z","iopub.execute_input":"2024-12-12T09:50:21.835954Z","iopub.status.idle":"2024-12-12T09:50:21.855545Z","shell.execute_reply.started":"2024-12-12T09:50:21.835906Z","shell.execute_reply":"2024-12-12T09:50:21.854511Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'gpu'\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU'\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:50:21.856738Z","iopub.execute_input":"2024-12-12T09:50:21.857008Z","iopub.status.idle":"2024-12-12T09:50:21.868136Z","shell.execute_reply.started":"2024-12-12T09:50:21.856970Z","shell.execute_reply":"2024-12-12T09:50:21.867388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor, StackingRegressor\n\ntabnet = TabNetWrapper(**TabNet_Params)\nxgboost = XGBRegressor(**XGB_Params)\nlight = lgb.LGBMRegressor(**Params, verbose=-1, n_estimators=300, random_state=SEED)\ncat = CatBoostRegressor(**CatBoost_Params)\n\nvoting = StackingRegressor(estimators=[\n    ('lightgbm', light),\n    ('xgboost', xgboost),\n    ('catboost', cat),\n    ('tabnet', tabnet)  # New:TabNet\n])\nSubmission, model, threshold = TrainML(voting, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:57:05.782300Z","iopub.execute_input":"2024-12-12T09:57:05.782666Z","iopub.status.idle":"2024-12-12T09:59:59.188003Z","shell.execute_reply.started":"2024-12-12T09:57:05.782634Z","shell.execute_reply":"2024-12-12T09:59:59.186952Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Debugging","metadata":{}},{"cell_type":"code","source":"X = train.drop(['sii'], axis=1)\ny = train['sii']\nprediction = threshold_Rounder(model.predict(X), threshold)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T10:01:16.843040Z","iopub.execute_input":"2024-12-12T10:01:16.843568Z","iopub.status.idle":"2024-12-12T10:01:21.746326Z","shell.execute_reply.started":"2024-12-12T10:01:16.843518Z","shell.execute_reply":"2024-12-12T10:01:21.745456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(prediction))\nX['prediction'] = prediction\nX['labels'] = y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T10:03:28.501963Z","iopub.execute_input":"2024-12-12T10:03:28.502309Z","iopub.status.idle":"2024-12-12T10:03:28.508685Z","shell.execute_reply.started":"2024-12-12T10:03:28.502278Z","shell.execute_reply":"2024-12-12T10:03:28.507538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X[y != prediction][X['labels'] == 0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T10:05:04.536517Z","iopub.execute_input":"2024-12-12T10:05:04.536889Z","iopub.status.idle":"2024-12-12T10:05:04.653624Z","shell.execute_reply.started":"2024-12-12T10:05:04.536856Z","shell.execute_reply":"2024-12-12T10:05:04.652491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nmissing0 = X[y != prediction][X['labels'] == 0]\npercentage = []\nlabels = []\nfor i in range(4):\n    percentage.append(len(missing0[missing0['prediction'] == i]))\n    labels.append(i)\nplt.pie(percentage, labels = labels)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T10:11:52.019751Z","iopub.execute_input":"2024-12-12T10:11:52.020448Z","iopub.status.idle":"2024-12-12T10:11:52.118433Z","shell.execute_reply.started":"2024-12-12T10:11:52.020411Z","shell.execute_reply":"2024-12-12T10:11:52.117400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"type = []\nlabels = []\nfor i in range(4):\n    type.append(missing0[missing0['prediction'] == i])\n    labels.append(i)\ntype[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T10:14:23.261380Z","iopub.execute_input":"2024-12-12T10:14:23.262336Z","iopub.status.idle":"2024-12-12T10:14:23.381483Z","shell.execute_reply.started":"2024-12-12T10:14:23.262283Z","shell.execute_reply":"2024-12-12T10:14:23.380556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(X[X['prediction'] == 0][X['labels'] == 0]['FGC-FGC_CU'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T10:18:28.640909Z","iopub.execute_input":"2024-12-12T10:18:28.641595Z","iopub.status.idle":"2024-12-12T10:18:28.898112Z","shell.execute_reply.started":"2024-12-12T10:18:28.641555Z","shell.execute_reply":"2024-12-12T10:18:28.897310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(type[1]['FGC-FGC_CU']) # Predict is 1 but true is 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T10:16:34.267850Z","iopub.execute_input":"2024-12-12T10:16:34.268227Z","iopub.status.idle":"2024-12-12T10:16:34.448632Z","shell.execute_reply.started":"2024-12-12T10:16:34.268194Z","shell.execute_reply":"2024-12-12T10:16:34.447650Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submit model","metadata":{}},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:53:23.022668Z","iopub.execute_input":"2024-12-12T09:53:23.023462Z","iopub.status.idle":"2024-12-12T09:53:23.030867Z","shell.execute_reply.started":"2024-12-12T09:53:23.023408Z","shell.execute_reply":"2024-12-12T09:53:23.029926Z"}},"outputs":[],"execution_count":null}]}