{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nn_splits = 5\nSEED = 42","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:48:18.363095Z","iopub.execute_input":"2024-12-12T12:48:18.363417Z","iopub.status.idle":"2024-12-12T12:48:23.694940Z","shell.execute_reply.started":"2024-12-12T12:48:18.363386Z","shell.execute_reply":"2024-12-12T12:48:23.693840Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Loading","metadata":{}},{"cell_type":"markdown","source":"## Load tabular dataset","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:48:23.696553Z","iopub.execute_input":"2024-12-12T12:48:23.697057Z","iopub.status.idle":"2024-12-12T12:48:23.761685Z","shell.execute_reply.started":"2024-12-12T12:48:23.697022Z","shell.execute_reply":"2024-12-12T12:48:23.760943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:48:23.762939Z","iopub.execute_input":"2024-12-12T12:48:23.763258Z","iopub.status.idle":"2024-12-12T12:48:23.923577Z","shell.execute_reply.started":"2024-12-12T12:48:23.763226Z","shell.execute_reply":"2024-12-12T12:48:23.922622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['id'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:48:23.926551Z","iopub.execute_input":"2024-12-12T12:48:23.927475Z","iopub.status.idle":"2024-12-12T12:48:23.934505Z","shell.execute_reply.started":"2024-12-12T12:48:23.927421Z","shell.execute_reply":"2024-12-12T12:48:23.933674Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load timeseries data","metadata":{}},{"cell_type":"markdown","source":"1. **process_file**: This function process file timeseries, extract general information in the file like count, mean, std, min, 25%, 50%, 75% and max of each features and then the features matrix is flattened to a vector to represent the data in the file\n2. **load_time_series** Format and load all timeseries files after processed in a folder.","metadata":{}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:48:23.935933Z","iopub.execute_input":"2024-12-12T12:48:23.936379Z","iopub.status.idle":"2024-12-12T12:48:23.944532Z","shell.execute_reply.started":"2024-12-12T12:48:23.936330Z","shell.execute_reply":"2024-12-12T12:48:23.943761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:48:23.945443Z","iopub.execute_input":"2024-12-12T12:48:23.945706Z","iopub.status.idle":"2024-12-12T12:49:48.210737Z","shell.execute_reply.started":"2024-12-12T12:48:23.945681Z","shell.execute_reply":"2024-12-12T12:49:48.209829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:49:48.211845Z","iopub.execute_input":"2024-12-12T12:49:48.212116Z","iopub.status.idle":"2024-12-12T12:49:48.271755Z","shell.execute_reply.started":"2024-12-12T12:49:48.212091Z","shell.execute_reply":"2024-12-12T12:49:48.270875Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Time series data is then merged to the tabular data","metadata":{}},{"cell_type":"code","source":"train = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:49:48.273135Z","iopub.execute_input":"2024-12-12T12:49:48.273478Z","iopub.status.idle":"2024-12-12T12:49:48.315837Z","shell.execute_reply.started":"2024-12-12T12:49:48.273424Z","shell.execute_reply":"2024-12-12T12:49:48.315063Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Here the KNN Imputer purpose is to fill the missing numerical values using 5 nearest neighbors. ","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers\n\n# Tạo một lớp GAN đơn giản để dự đoán giá trị bị thiếu\nclass GANImputer:\n    def __init__(self, latent_dim=100, epochs=1000, batch_size=32):\n        self.latent_dim = latent_dim\n        self.epochs = epochs\n        self.batch_size = batch_size\n        self.generator = self.build_generator()\n        self.discriminator = self.build_discriminator()\n        self.gan = self.build_gan()\n\n    def build_generator(self):\n        model = tf.keras.Sequential([\n            layers.Dense(128, activation='relu', input_dim=self.latent_dim),\n            layers.Dense(256, activation='relu'),\n            layers.Dense(512, activation='relu'),\n            layers.Dense(1, activation='linear')\n        ])\n        return model\n\n    def build_discriminator(self):\n        model = tf.keras.Sequential([\n            layers.Dense(512, activation='relu', input_dim=1),\n            layers.Dense(256, activation='relu'),\n            layers.Dense(128, activation='relu'),\n            layers.Dense(1, activation='sigmoid')\n        ])\n        model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n        return model\n\n    def build_gan(self):\n        self.discriminator.trainable = False\n        model = tf.keras.Sequential([\n            self.generator,\n            self.discriminator\n        ])\n        model.compile(optimizer='adam', loss='binary_crossentropy')\n        return model\n\n    def train(self, data):\n        if data.shape[0] < self.batch_size:\n            raise ValueError(\"Not enough data to train the GAN. Ensure the batch size is smaller than the number of samples.\")\n        \n        valid = np.ones((self.batch_size, 1))\n        fake = np.zeros((self.batch_size, 1))\n\n        for epoch in range(self.epochs):\n            # Lấy mẫu dữ liệu thật\n            idx = np.random.randint(0, data.shape[0], self.batch_size)\n            real_data = data[idx]\n\n            # Sinh dữ liệu giả\n            noise = np.random.normal(0, 1, (self.batch_size, self.latent_dim))\n            generated_data = self.generator.predict(noise)\n\n            # Huấn luyện Discriminator\n            d_loss_real = self.discriminator.train_on_batch(real_data, valid)\n            d_loss_fake = self.discriminator.train_on_batch(generated_data, fake)\n            d_loss = 0.5 * np.add(d_loss_real, d_loss_fake)\n\n            # Huấn luyện Generator\n            noise = np.random.normal(0, 1, (self.batch_size, self.latent_dim))\n            g_loss = self.gan.train_on_batch(noise, valid)\n\n            if epoch % 100 == 0:\n                print(f\"Epoch {epoch}/{self.epochs} - D Loss: {d_loss[0]:.4f}, G Loss: {g_loss:.4f}\")\n\n    def impute(self, data):\n        missing_idx = np.isnan(data)\n        complete_data = data.copy()\n        \n        for i in range(data.shape[1]):  # Với từng cột\n            missing_rows = missing_idx[:, i]\n            if np.any(missing_rows):\n                # Lấy noise để dự đoán giá trị thiếu\n                noise = np.random.normal(0, 1, (missing_rows.sum(), self.latent_dim))\n                generated_values = self.generator.predict(noise)\n                complete_data[missing_rows, i] = generated_values[:, 0]\n        return complete_data\n\n# Tiền xử lý dữ liệu\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nnumeric_data = train[numeric_cols].values\n\n# Loại bỏ hàng chứa toàn giá trị NaN\nnon_nan_rows = ~np.isnan(numeric_data).all(axis=1)\nnumeric_data = numeric_data[non_nan_rows]\n\n# Chuẩn bị GANs\ngan_imputer = GANImputer(epochs=500, batch_size=64)\n\n# # Tách dữ liệu không bị thiếu\n# non_missing_data = numeric_data[~np.isnan(numeric_data).any(axis=1)]\n\n# # Kiểm tra đủ dữ liệu để huấn luyện\n# if non_missing_data.shape[0] > 0:\n#     # Huấn luyện GAN trên dữ liệu không thiếu\n#     gan_imputer.train(non_missing_data)\n# else:\n#     raise ValueError(\"No complete rows available for training GAN.\")\n\n# Điền giá trị bị thiếu\nimputed_data = gan_imputer.impute(numeric_data)\n\n# Chuyển thành DataFrame\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n\n# Làm tròn cột 'sii'\nif 'sii' in train_imputed.columns:\n    train_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\n# Gán lại các cột không phải số\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\ntrain = train_imputed\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:55:45.091345Z","iopub.execute_input":"2024-12-12T12:55:45.091753Z","iopub.status.idle":"2024-12-12T12:56:17.488633Z","shell.execute_reply.started":"2024-12-12T12:55:45.091723Z","shell.execute_reply":"2024-12-12T12:56:17.487826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:56:43.693690Z","iopub.execute_input":"2024-12-12T12:56:43.694047Z","iopub.status.idle":"2024-12-12T12:56:43.706470Z","shell.execute_reply.started":"2024-12-12T12:56:43.694019Z","shell.execute_reply":"2024-12-12T12:56:43.705359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:56:46.489930Z","iopub.execute_input":"2024-12-12T12:56:46.490270Z","iopub.status.idle":"2024-12-12T12:56:46.579890Z","shell.execute_reply.started":"2024-12-12T12:56:46.490241Z","shell.execute_reply":"2024-12-12T12:56:46.578949Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Filtering","metadata":{}},{"cell_type":"markdown","source":"Select the columns which is present in test data to train","metadata":{}},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:02.634731Z","iopub.execute_input":"2024-12-12T12:57:02.635093Z","iopub.status.idle":"2024-12-12T12:57:02.643822Z","shell.execute_reply.started":"2024-12-12T12:57:02.635063Z","shell.execute_reply":"2024-12-12T12:57:02.642869Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Drop the NaN sii value","metadata":{}},{"cell_type":"code","source":"train = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:04.837728Z","iopub.execute_input":"2024-12-12T12:57:04.838803Z","iopub.status.idle":"2024-12-12T12:57:04.856742Z","shell.execute_reply.started":"2024-12-12T12:57:04.838735Z","shell.execute_reply":"2024-12-12T12:57:04.855849Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Fill the missing categorical data with \"Missing\" ","metadata":{}},{"cell_type":"code","source":"cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:06.961805Z","iopub.execute_input":"2024-12-12T12:57:06.962667Z","iopub.status.idle":"2024-12-12T12:57:06.993574Z","shell.execute_reply.started":"2024-12-12T12:57:06.962627Z","shell.execute_reply":"2024-12-12T12:57:06.992599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Create a mapping from string to integer to push data to the model (Use one hot encode instead)","metadata":{}},{"cell_type":"code","source":"def create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:09.082143Z","iopub.execute_input":"2024-12-12T12:57:09.082915Z","iopub.status.idle":"2024-12-12T12:57:09.125956Z","shell.execute_reply.started":"2024-12-12T12:57:09.082874Z","shell.execute_reply":"2024-12-12T12:57:09.124848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:12.149705Z","iopub.execute_input":"2024-12-12T12:57:12.150491Z","iopub.status.idle":"2024-12-12T12:57:12.169709Z","shell.execute_reply.started":"2024-12-12T12:57:12.150454Z","shell.execute_reply":"2024-12-12T12:57:12.168484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:14.181910Z","iopub.execute_input":"2024-12-12T12:57:14.182265Z","iopub.status.idle":"2024-12-12T12:57:14.262639Z","shell.execute_reply.started":"2024-12-12T12:57:14.182234Z","shell.execute_reply":"2024-12-12T12:57:14.261621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sii'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:17.195704Z","iopub.execute_input":"2024-12-12T12:57:17.196081Z","iopub.status.idle":"2024-12-12T12:57:17.204150Z","shell.execute_reply.started":"2024-12-12T12:57:17.196051Z","shell.execute_reply":"2024-12-12T12:57:17.202718Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training Function","metadata":{}},{"cell_type":"markdown","source":"**quadratic_weighted_kappa**: calculate QWK value","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:19.707452Z","iopub.execute_input":"2024-12-12T12:57:19.707928Z","iopub.status.idle":"2024-12-12T12:57:19.712617Z","shell.execute_reply.started":"2024-12-12T12:57:19.707880Z","shell.execute_reply":"2024-12-12T12:57:19.711616Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**threshold_Rounder**: Turn the sii from PCIAT_Total to categorical ","metadata":{}},{"cell_type":"code","source":"def threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:23.015347Z","iopub.execute_input":"2024-12-12T12:57:23.016032Z","iopub.status.idle":"2024-12-12T12:57:23.020843Z","shell.execute_reply.started":"2024-12-12T12:57:23.015990Z","shell.execute_reply":"2024-12-12T12:57:23.019766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**evaluate_predictions**: this function evaluate the prediction of the model by first turn integer prediction values to categorical values and then calculate QWK from it and the true labels.","metadata":{}},{"cell_type":"code","source":"def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:24.819463Z","iopub.execute_input":"2024-12-12T12:57:24.820484Z","iopub.status.idle":"2024-12-12T12:57:24.824835Z","shell.execute_reply.started":"2024-12-12T12:57:24.820435Z","shell.execute_reply":"2024-12-12T12:57:24.823867Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**TrainML**: Train the model using K-Fold, The model is regression model, predict a real value represent how bad the patient was. The value may not explicitly different, so we re-define the threshold to make it split more accurate","metadata":{}},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    # Apply K-Fold\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        # Train model\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        # Round to integer values\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        #Predict with test dataset\n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Using optimizer to find the best threshold\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    # Use the threshold retrive from the optimizer to predict again to evaluate\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    # Use the threshold retrive from the optimizer to predict test\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    # Create submition\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission,model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:31.391859Z","iopub.execute_input":"2024-12-12T12:57:31.392214Z","iopub.status.idle":"2024-12-12T12:57:31.403316Z","shell.execute_reply.started":"2024-12-12T12:57:31.392184Z","shell.execute_reply":"2024-12-12T12:57:31.402362Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create model and train the model","metadata":{}},{"cell_type":"markdown","source":"We first try to train a model to take it as baseline. The baseline here is LGBM, which is an Gradient Boosting framework","metadata":{}},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n}\n\nLight = lgb.LGBMRegressor(**Params, verbose=-1, n_estimators=200, random_state=SEED)\nSubmission,model = TrainML(Light,test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:34.639402Z","iopub.execute_input":"2024-12-12T12:57:34.640130Z","iopub.status.idle":"2024-12-12T12:57:49.555108Z","shell.execute_reply.started":"2024-12-12T12:57:34.640095Z","shell.execute_reply":"2024-12-12T12:57:49.553961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submit model","metadata":{}},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:57:52.492437Z","iopub.execute_input":"2024-12-12T12:57:52.492800Z","iopub.status.idle":"2024-12-12T12:57:52.510098Z","shell.execute_reply.started":"2024-12-12T12:57:52.492754Z","shell.execute_reply":"2024-12-12T12:57:52.509116Z"}},"outputs":[],"execution_count":null}]}