{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:12:22.079672Z","iopub.execute_input":"2024-11-28T04:12:22.080031Z","iopub.status.idle":"2024-11-28T04:12:22.084058Z","shell.execute_reply.started":"2024-11-28T04:12:22.080002Z","shell.execute_reply":"2024-11-28T04:12:22.082972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, mean_squared_error\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.decomposition import PCA\nfrom sklearn.datasets import make_classification\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom scipy.stats import skew\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n# from pytorch_tabnet.tab_model import TabNetRegressor\n# from pytorch_tabnet.callbacks import Callback\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor, early_stopping\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nimport matplotlib.pyplot as plt\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-30T05:30:04.948859Z","iopub.execute_input":"2024-11-30T05:30:04.949265Z","iopub.status.idle":"2024-11-30T05:30:25.891373Z","shell.execute_reply.started":"2024-11-30T05:30:04.949214Z","shell.execute_reply":"2024-11-30T05:30:25.890015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\ndef print_isnan(data):\n    print(train_ts_encoded.isna().sum())\n\n    \ndef print_isinf(data):\n    inf_counts = data.isin([float('inf'), float('-inf')]).sum()\n    print(inf_counts)\n\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n    \ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:30:25.893964Z","iopub.execute_input":"2024-11-30T05:30:25.894701Z","iopub.status.idle":"2024-11-30T05:30:25.911961Z","shell.execute_reply.started":"2024-11-30T05:30:25.894661Z","shell.execute_reply":"2024-11-30T05:30:25.9106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\n\n# unrealistic weight\ntrain['Physical-Weight'].replace(0, np.nan, inplace=True)\ntest['Physical-Weight'].replace(0, np.nan, inplace=True)\n\nprint(\"train ts process, with stats of each column -> originally with 104 columns for auto-encoding\")\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n# no null or inf values\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded['id'] = test_ts[\"id\"]","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:30:25.913556Z","iopub.execute_input":"2024-11-30T05:30:25.913908Z","iopub.status.idle":"2024-11-30T05:32:17.227765Z","shell.execute_reply.started":"2024-11-30T05:30:25.913876Z","shell.execute_reply":"2024-11-30T05:32:17.225897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the train_ts has only 996 rows, the rest unmatched train have NaN values\nprint(\"w/o ts feats:\", train.shape, test.shape)\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\nprint(\"w ts feats:\", train.shape, test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:32:17.232336Z","iopub.execute_input":"2024-11-30T05:32:17.234522Z","iopub.status.idle":"2024-11-30T05:32:17.278372Z","shell.execute_reply.started":"2024-11-30T05:32:17.234472Z","shell.execute_reply":"2024-11-30T05:32:17.277036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 0 for None, 1 for Mild, 2 for Moderate, and 3 for Severe\n# todo: calibration or sampling technique\nprint(\"train total:\", len(train), \"isnan:\", sum(train['sii'].isna()))\ntrain.groupby('sii')['id'].size()","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:32:17.279885Z","iopub.execute_input":"2024-11-30T05:32:17.280271Z","iopub.status.idle":"2024-11-30T05:32:17.299102Z","shell.execute_reply.started":"2024-11-30T05:32:17.2802Z","shell.execute_reply":"2024-11-30T05:32:17.297886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(train.shape, test.shape)\n# test_cols = set(test.columns)\n# train_cols = set(train.columns)\n\n# train_ids = set(train.id)\n# test_ids = set(test.id)\n\n# print(\"extra columns\", sorted(train_cols - test_cols))\n# print(test_ids - train_ids)\n# train.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:32:17.300787Z","iopub.execute_input":"2024-11-30T05:32:17.301273Z","iopub.status.idle":"2024-11-30T05:32:17.307579Z","shell.execute_reply.started":"2024-11-30T05:32:17.301207Z","shell.execute_reply":"2024-11-30T05:32:17.306016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train[\"SDS-SDS_Total_T\"].plot.hist()","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:32:17.309273Z","iopub.execute_input":"2024-11-30T05:32:17.309696Z","iopub.status.idle":"2024-11-30T05:32:17.321381Z","shell.execute_reply.started":"2024-11-30T05:32:17.309647Z","shell.execute_reply":"2024-11-30T05:32:17.319799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# age_gp = train.groupby([\"BMI_Category\", \"sii\"])[\"id\"].count() / train.groupby([\"BMI_Category\"])[\"id\"].size()\n\n# print(age_gp)","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:32:17.322925Z","iopub.execute_input":"2024-11-30T05:32:17.323448Z","iopub.status.idle":"2024-11-30T05:32:17.334562Z","shell.execute_reply.started":"2024-11-30T05:32:17.323393Z","shell.execute_reply":"2024-11-30T05:32:17.333082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feat Engineering","metadata":{}},{"cell_type":"code","source":"from typing import List\n\nbase_feats = [x for x in test.columns if x != 'id' and not x.endswith('Season')]\n# season_feats = [x for x in test.columns if x.endswith('Season')]\nseason_feats = []  # disable season feats\n\n\ncat_feats = []\ndef feat_eng(df: pd.DataFrame, \n             stage: str='train', \n             df_age_feats: pd.DataFrame=None,\n             df_sex_feats: pd.DataFrame=None,\n             df_int_feats: pd.DataFrame=None,\n             df_bmi_feats: pd.DataFrame=None,\n            ):\n    global cat_feats\n    cat_feats = season_feats + ['Basic_Demos-Sex', 'Age_Group', 'BMI_Category',\n                                'Blood_Pressure_Cat']\n    def _categorize_bp(row):\n        if row['Physical-Systolic_BP'] < 120 and row['Physical-Diastolic_BP'] < 80:\n            return 0\n        elif 120 <= row['Physical-Systolic_BP'] < 140 or 80 <= row['Physical-Diastolic_BP'] < 90:\n            return 1\n        else:\n            return 2\n    \n    df = df.copy()\n    # todo impute questionaires based on simularity\n    feats_all = base_feats + season_feats\n    \n    df['Enroll_Summer'] = df['Basic_Demos-Enroll_Season'].apply(lambda x: 1 if x == 'Summer' else 0)\n    feats_all.append('Enroll_Summer')\n    \n    df['Age_Group'] = pd.cut(\n        df['Basic_Demos-Age'], \n        bins=[0, 6, 8, 10, 12, 14, 16, 18, np.inf],\n        labels=range(8)\n    )\n    feats_all.append('Age_Group')\n    \n    df['BMI_Category'] = pd.cut(df['Physical-BMI'], bins=[0, 18.5, 24.9, 29.9, np.inf], labels=range(4))\n    feats_all.append('BMI_Category')\n    \n    df['Blood_Pressure_Cat'] = df.apply(_categorize_bp, axis=1)\n    feats_all.append('Blood_Pressure_Cat')\n    \n    # Height/Weight Ratio\n    df['Height_Weight_Ratio'] = df['Physical-Height'] / df['Physical-Weight']\n    # Endurance Time in Minutes\n    df['Endurance_Time'] = df['Fitness_Endurance-Time_Mins'] + df['Fitness_Endurance-Time_Sec'] / 60\n    df['Grip_Strength_Total'] = df['FGC-FGC_GSND'] + df['FGC-FGC_GSD']\n    # Flexibility (Sit & Reach) Balance - Difference between left and right\n    df['Flexibility_Balance'] = df['FGC-FGC_SRL'] - df['FGC-FGC_SRR']\n    df['General_Fitness_Score'] = df['Grip_Strength_Total'] + df['Endurance_Time'] + df['FGC-FGC_PU']\n    # Calculate Body Composition Profile as a weighted score\n    df['Body_Composition_Score'] = (\n        0.4 * df['BIA-BIA_Fat'] + \n        0.3 * df['BIA-BIA_FFM'] + \n        0.3 * df['BIA-BIA_SMM']\n    )\n    # Activity Summary Score - combining adolescent and child scores (fill NaN with 0)\n    df['Activity_Summary_Score'] = df['PAQ_A-PAQ_A_Total'].fillna(0) + df['PAQ_C-PAQ_C_Total'].fillna(0)\n    # Create a binary feature for 'poor sleep' if SDS_Total_T indicates a high sleep disturbance level (e.g., T > 60 as a threshold)\n    df['Poor_Sleep'] = df['SDS-SDS_Total_T'].apply(lambda x: 1 if x > 60 else 0)\n    feats_all.extend([\n        'Height_Weight_Ratio', 'Endurance_Time', 'Grip_Strength_Total',\n        'Flexibility_Balance', 'General_Fitness_Score', 'Body_Composition_Score',\n        'Activity_Summary_Score', 'Poor_Sleep'\n    ])\n    \n    # basic features 2\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    feats_all.extend([\n        'BMI_Age', 'Internet_Hours_Age', 'BMI_Internet_Hours',\n        'BFP_BMI', 'FFMI_BFP', 'FMI_BFP',\n        'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n        'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW'\n    ])\n    \n    if stage == 'train':\n        df_age_feats = encode_cat_features(df, 'Age_Group')\n        df_sex_feats = encode_cat_features(df, 'Basic_Demos-Sex')\n        df_int_feats = encode_cat_features(df, \"PreInt_EduHx-computerinternet_hoursday\")\n        df_bmi_feats = encode_cat_features(df, \"BMI_Category\")\n              \n    age_feats = [x for x in df_age_feats.columns if x != 'Age_Group']\n    df = df.merge(df_age_feats, on='Age_Group', how='left')\n    feats_all.extend(age_feats)\n\n    sex_feats = [x for x in df_sex_feats.columns if x != 'Basic_Demos-Sex']\n    df = df.merge(df_sex_feats, on='Basic_Demos-Sex', how='left')\n    feats_all.extend(sex_feats)\n\n    int_feats = [x for x in df_int_feats.columns if x != \"PreInt_EduHx-computerinternet_hoursday\"]\n    df = df.merge(df_int_feats, on=\"PreInt_EduHx-computerinternet_hoursday\", how='left')\n    feats_all.extend(int_feats)\n\n    bmi_feats = [x for x in df_bmi_feats.columns if x != \"BMI_Category\"]\n    df = df.merge(df_bmi_feats, on=\"BMI_Category\", how='left')\n    feats_all.extend(bmi_feats)\n    \n#     print(\"Process actigraphy features ...\")\n#     feats_all.extend(time_series_cols)\n#     dname = f'/kaggle/input/child-mind-institute-problematic-internet-use/series_{stage}.parquet'\n#     df_ts, ts_feat = process_actigraphy_features(dname)\n    \n#     df = df.merge(df_ts, on='id', how='left')\n#     feats_all.extend(ts_feat)\n\n    season_mapping = {\n        \"Spring\": 1,\n        \"Summer\": 2,\n        \"Fall\": 3,\n        \"Winter\": 4\n    }\n\n    for f in season_feats:\n        df[f] = df[f].map(season_mapping)\n\n    for f in cat_feats:\n        df[f] = df[f].astype('category')\n    return df[feats_all], df_age_feats, df_sex_feats, df_int_feats, df_bmi_feats\n\n\ndef add_age_features(df: pd.DataFrame) -> pd.DataFrame:\n    prob = df.groupby(['Age_Group', 'sii'])['id'].size() / df.groupby(['Age_Group'])['id'].size()\n    df_age_feats = prob.unstack(level='sii')\n    age_sii_feats = [f'age_sii_{int(col)}' for col in df_age_feats.columns]\n    df_age_feats.columns = age_sii_feats\n    df_age_feats.reset_index(inplace=True)\n\n    # PCIAT columns\n    pciat_cols = [f\"PCIAT-PCIAT_0{str(x)}\" if len(str(x)) == 1 else f\"PCIAT-PCIAT_{str(x)}\" for x in range(1, 21)] + ['PCIAT-PCIAT_Total']\n    df_age_pciat = df.groupby(['Age_Group'])[pciat_cols].mean()\n    pciat_feats = [f\"{c}_mean\" for c in pciat_cols]\n    df_age_pciat.columns = pciat_feats\n    df_age_pciat.reset_index(inplace=True)\n\n    # df age feats has all features related to age group\n    df_age_feats = df_age_feats.merge(df_age_pciat, on='Age_Group')\n    return df_age_feats\n\ndef encode_cat_features(df: pd.DataFrame, col_name: str) -> pd.DataFrame:\n    \"\"\"\n    column_name is the existing categorical column name, doing categorical target encoding\n    \"\"\"\n    prob = df.groupby([col_name, 'sii'])['id'].size() / df.groupby([col_name])['id'].size()\n    df_cat_feats = prob.unstack(level='sii')\n    cat_sii_feats = [f'{col_name}_sii_{int(col)}' for col in df_cat_feats.columns]\n    df_cat_feats.columns = cat_sii_feats\n    df_cat_feats.reset_index(inplace=True)\n    \n    agg_dict = {\n        f\"{col_name}_pciat_mean\": ('PCIAT-PCIAT_Total', np.mean),\n        f\"{col_name}_pciat_min\": ('PCIAT-PCIAT_Total', np.min),\n        f'{col_name}_pciat_max': ('PCIAT-PCIAT_Total', np.max),\n        f'{col_name}_pciat_skew': ('PCIAT-PCIAT_Total', lambda x: skew(x.dropna()))\n    }\n    df_cat_pciat = df.groupby(col_name).agg(**agg_dict).reset_index()\n\n    df_cat_feats = df_cat_feats.merge(df_cat_pciat, on=col_name)\n    return df_cat_feats\n\ndef _single_process(dirname, filename):\n    participant_id = str(filename.split('=')[1])\n    \n    full_path = os.path.join(dirname, filename, 'part-0.parquet')\n    actigraphy = pd.read_parquet(full_path)\n    \n    # Feature Aggregation\n    participant_features = {\n        'id': participant_id,\n        'Avg_ENMO': actigraphy['enmo'].mean(),\n        'Max_ENMO': actigraphy['enmo'].max(),\n        'Sedentary_Time': (actigraphy['enmo'] == 0).sum() * 5,  # assuming each row is 5-second interval\n        'Total_Activity': (actigraphy['enmo'] > 0).sum() * 5,\n        'Avg_Light': actigraphy['light'].mean(),\n        'NonWear_Time': actigraphy['non-wear_flag'].sum() * 5\n    }\n    return participant_features\n\n\ndef process_actigraphy_features(dirname: str) -> (pd.DataFrame, List):\n    \"\"\"\n    Return time series data frame and features\n    \"\"\"\n    with ThreadPoolExecutor(max_workers=5) as executor:\n        results = list(executor.map(lambda fname: _single_process(dirname, fname), os.listdir(dirname)))\n    df = pd.DataFrame(results)\n    feat = [t for t in df.columns if t != 'id']\n    return df, feat","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:32:17.336722Z","iopub.execute_input":"2024-11-30T05:32:17.337096Z","iopub.status.idle":"2024-11-30T05:32:17.390042Z","shell.execute_reply.started":"2024-11-30T05:32:17.337063Z","shell.execute_reply":"2024-11-30T05:32:17.38858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def impute_data(df: pd.DataFrame):\n    print(\"Impute ...\")\n    imputer = KNNImputer(n_neighbors=5)\n    numeric_cols = df.select_dtypes(include=['int32', 'int64', 'float64']).columns\n    imputed_data = imputer.fit_transform(df[numeric_cols])\n\n    train_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n    train_imputed['sii'] = train_imputed['sii'].round().astype(int)\n    for col in df.columns:\n        if col not in numeric_cols:\n            train_imputed[col] = train[col]\n    return train_imputed\n        \ntrain_imputed = impute_data(train)\n\nprint(\"Feat engineering\")\ntrain_feats, df_age_feats, df_sex_feats, df_int_feats, df_bmi_feats = feat_eng(train_imputed, stage='train')\ntest_feats, _, _, _, _ = feat_eng(\n    test, \n    stage='test', \n    df_age_feats=df_age_feats,\n    df_sex_feats=df_sex_feats,\n    df_int_feats=df_int_feats,\n    df_bmi_feats=df_bmi_feats\n)\n\nprint(train_feats.shape, test_feats.shape)\n\n# print(\"Impute train ...\")\n# train_feats['sii'] = y\n# numeric_cols = train_feats.select_dtypes(include=['int32', 'int64', 'float64', 'int64']).columns\n# print(\"len(numeric_cols):\", len(numeric_cols))\n\n# imputer = KNNImputer(n_neighbors=5)\n# imputed_data = imputer.fit_transform(train_feats[numeric_cols])\n# train_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n# train_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\n# print(\"Impute test ...\")\n# test_feats['sii'] = np.nan\n# imputed_data = imputer.transform(test_feats[numeric_cols])\n# test_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n\n# for col in train_feats.columns:\n#     if col not in numeric_cols:\n#         train_imputed[col] = train_feats[col]\n#         test_imputed[col] = test_feats[col]\n\n# y = train_imputed['sii']\n# imputer_pred = test_imputed['sii'].round().astype(int)\n\n# train_feats = train_imputed.drop(['sii'], axis=1)\n# test_feats = test_imputed[train_feats.columns]\n# print(train_feats.shape, len(y), test_feats.shape)","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:32:17.393435Z","iopub.execute_input":"2024-11-30T05:32:17.393853Z","iopub.status.idle":"2024-11-30T05:32:26.456815Z","shell.execute_reply.started":"2024-11-30T05:32:17.393817Z","shell.execute_reply":"2024-11-30T05:32:26.455418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_age_feats.to_csv(\"df_age_feats.csv\", index=False)\ndf_sex_feats.to_csv(\"df_sex_feats.csv\", index=False)\ndf_int_feats.to_csv(\"df_int_feats.csv\", index=False)\ndf_bmi_feats.to_csv(\"df_bmi_feats.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-30T05:34:00.259301Z","iopub.execute_input":"2024-11-30T05:34:00.259703Z","iopub.status.idle":"2024-11-30T05:34:00.271806Z","shell.execute_reply.started":"2024-11-30T05:34:00.259667Z","shell.execute_reply":"2024-11-30T05:34:00.270603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exit()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\nmetrics = pd.DataFrame(np.zeros((5, 1)), columns=['val_kappa'], index=[f\"fold_{str(i)}\" for i in range(n_splits)])\nSKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\ntest_pred = np.zeros(len(test_feats))\n# feature_imp = pd.DataFrame(np.zeros((len(train_feats.columns), 1)), \n#                            columns=['importance'], \n#                            index=train_feats.columns)\n\n\n# y = train['sii']\n# y_filled = y.fillna(-1)\n\nX = train_feats\ny = train_imputed['sii']\n\nfor fold, (train_idx, val_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n    train_X, train_y = X.iloc[train_idx], y.iloc[train_idx]\n    val_X, val_y = X.iloc[val_idx], y.iloc[val_idx]\n\n#     print(\"Impute train ...\")\n#     train_X = impute_data(train_X)\n#     train_feats_X, df_age_feats, df_sex_feats, df_int_feats, df_bmi_feats = feat_eng(train_X, stage='train')\n#     val_feats_X, _, _, _, _ = feat_eng(\n#         val_X, \n#         stage='test', \n#         df_age_feats=df_age_feats,\n#         df_sex_feats=df_sex_feats,\n#         df_int_feats=df_int_feats,\n#         df_bmi_feats=df_bmi_feats\n#     )\n    train_feats_X, val_feats_X = train_X, val_X\n    print(\"train shape:\", train_feats_X.shape, \"val shape:\", val_feats_X.shape)\n            \n    lgb_params = {\n        \"objective\": \"mae\",\n        \"n_estimators\": 1000,\n        \"num_leaves\": 20,\n        \"subsample\": 0.8,\n        \"colsample_bytree\": 0.7,\n        \"learning_rate\": 0.008,\n        \"n_jobs\": 4,\n        \"device\": \"gpu\",\n        \"verbosity\": -1,\n        \"importance_type\": \"gain\",\n        \"max_depth\": 4,\n        \"reg_alpha\": 6,\n        \"reg_lambda\": 0.08,\n    }\n    \n    model = LGBMRegressor(**lgb_params)\n    print(\"train_y nan:\", sum(train_y.isna()))\n    model.fit(\n        train_feats_X, \n        train_y, \n        categorical_feature=cat_feats,\n        eval_set=[(val_feats_X, val_y)],\n        callbacks=[\n            early_stopping(stopping_rounds=30),\n        ]\n    )\n#     feature_imp['importance'] += model.feature_importances_ / n_splits\n    \n    y_pred = model.predict(val_feats_X).round()\n    val_y = val_y.reset_index(drop=True)\n    mask = val_y.notna()\n    val_kappa = quadratic_weighted_kappa(val_y[mask], y_pred[mask])\n    \n    test_pred += model.predict(test_feats) / n_splits\n    \n    # Store the kappa score for this fold\n    metrics.loc[f\"fold_{fold}\", 'val_kappa'] = val_kappa\n    print(f\"Fold {fold + 1} Kappa: {val_kappa}\")\n\ntest_pred = test_pred.round(0).clip(0, 3)\nprint(metrics)\nprint(\"Average Kappa Score:\", metrics['val_kappa'].mean())\n","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:48:35.537857Z","iopub.execute_input":"2024-11-28T04:48:35.538223Z","iopub.status.idle":"2024-11-28T04:48:53.628139Z","shell.execute_reply.started":"2024-11-28T04:48:35.538191Z","shell.execute_reply":"2024-11-28T04:48:53.626923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# full_train(with NA) + 5 fold + base agg feats: cv: 0.299 -> LB: 0.302\n# full_train(with NA) + ts_auto-encoder + 5 fold + base agg feats: cv: 0.312 -> LB: 0.33\n\nsubmission = pd.DataFrame({\n    'id': sample['id'],\n    'sii': test_pred\n})\n\nprint(submission.head())\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:48:59.70451Z","iopub.execute_input":"2024-11-28T04:48:59.705002Z","iopub.status.idle":"2024-11-28T04:48:59.713549Z","shell.execute_reply.started":"2024-11-28T04:48:59.704965Z","shell.execute_reply":"2024-11-28T04:48:59.712614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_feats.columns == val_feats_X.columns","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:34:24.789701Z","iopub.execute_input":"2024-11-28T04:34:24.790473Z","iopub.status.idle":"2024-11-28T04:34:24.796356Z","shell.execute_reply.started":"2024-11-28T04:34:24.790436Z","shell.execute_reply":"2024-11-28T04:34:24.795517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exit()","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:16.332875Z","iopub.execute_input":"2024-11-28T04:14:16.333302Z","iopub.status.idle":"2024-11-28T04:14:16.342857Z","shell.execute_reply.started":"2024-11-28T04:14:16.333254Z","shell.execute_reply":"2024-11-28T04:14:16.341782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['int32', 'int64', 'float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:16.346422Z","iopub.execute_input":"2024-11-28T04:14:16.34684Z","iopub.status.idle":"2024-11-28T04:14:23.18681Z","shell.execute_reply.started":"2024-11-28T04:14:16.346793Z","shell.execute_reply":"2024-11-28T04:14:23.185431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop('id', axis=1)\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.18744Z","iopub.status.idle":"2024-11-28T04:14:23.187768Z","shell.execute_reply.started":"2024-11-28T04:14:23.18761Z","shell.execute_reply":"2024-11-28T04:14:23.187626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop('id', axis=1)\ntest","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.189349Z","iopub.status.idle":"2024-11-28T04:14:23.189664Z","shell.execute_reply.started":"2024-11-28T04:14:23.189512Z","shell.execute_reply":"2024-11-28T04:14:23.189528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.191023Z","iopub.status.idle":"2024-11-28T04:14:23.191296Z","shell.execute_reply.started":"2024-11-28T04:14:23.191163Z","shell.execute_reply":"2024-11-28T04:14:23.191177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.192473Z","iopub.status.idle":"2024-11-28T04:14:23.192739Z","shell.execute_reply.started":"2024-11-28T04:14:23.192607Z","shell.execute_reply":"2024-11-28T04:14:23.192621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.19354Z","iopub.status.idle":"2024-11-28T04:14:23.193899Z","shell.execute_reply.started":"2024-11-28T04:14:23.193704Z","shell.execute_reply":"2024-11-28T04:14:23.19372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.195343Z","iopub.status.idle":"2024-11-28T04:14:23.195661Z","shell.execute_reply.started":"2024-11-28T04:14:23.195516Z","shell.execute_reply":"2024-11-28T04:14:23.195532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.19661Z","iopub.status.idle":"2024-11-28T04:14:23.196915Z","shell.execute_reply.started":"2024-11-28T04:14:23.196745Z","shell.execute_reply":"2024-11-28T04:14:23.196778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tab_model.pt'\n        \n    def fit(self, X, y):\n        X_imputed = self.imputer.fit_transform(X)\n            \n        if hasattr(y, 'values'):\n            y = y.values\n\n        X_train, X_valid, y_train, y_valid = train_test_split(X_imputed, y, test_size=0.2, random_state=SEED)\n\n        history = self.model.fit(\n            X_train = X_train,\n            y_train = y_train.reshape(-1, 1),\n            eval_set = [(X_valid, y_valid.reshape(-1, 1))],\n            eval_name = ['valid'],\n            eval_metric = ['mse'],\n            max_epochs = 500,\n            patience = 50,\n            batch_size = 1024,\n            virtual_batch_size = 128,\n            num_workers = 0,\n            drop_last = False,\n            callbacks = [\n                TabNetPretrainedModelCheckpoint(\n                    filepath = self.best_model_path,\n                    monitor = 'valid_mse',\n                    save_best_only = True,\n                    verbose = True\n                )\n            ]\n        )\n\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)\n\n        return self\n\n    def predict(self, X):\n        X_imputed = self.imputer.fit_transform(X)\n        return self.model.predict(X_imputed).flatten()\n\n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = self.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n\n        return result\n\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', save_best_only=True, verbose=1):\n        super().__init__()\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        if (self.mode == 'min' and current < self.best) or (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.198014Z","iopub.status.idle":"2024-11-28T04:14:23.198288Z","shell.execute_reply.started":"2024-11-28T04:14:23.198151Z","shell.execute_reply":"2024-11-28T04:14:23.198166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ObliqueDecisionTreeNode:\n    def __init__(self, depth=0, max_depth=5, alpha=0.01, min_samples_split=2, min_samples_leaf=1):\n        self.left = None\n        self.right = None\n        self.depth = depth\n        self.max_depth = max_depth\n        self.alpha = alpha\n        self.min_samples_split = min_samples_split\n        self.min_samples_leaf = min_samples_leaf\n        self.is_leaf = False\n        self.coefficients = None\n        self.bias = None\n        self.prediction = None\n\n    def fit(self, X, y):\n        if self.depth >= self.max_depth or len(y) < self.min_samples_split:\n            self.is_leaf = True\n            self.prediction = np.mean(y)\n            return\n\n        def objective(params):\n            coefficients, bias = params[:-1], params[-1]\n            predictions = X @ coefficients + bias\n            left_mask = predictions < 0\n            right_mask = ~left_mask\n            \n            if np.sum(left_mask) == 0 or np.sum(right_mask) == 0:\n                return float('inf')\n\n            left_variance = np.var(y[left_mask]) if np.sum(left_mask) > 1 else 0\n            right_variance = np.var(y[right_mask]) if np.sum(right_mask) > 1 else 0\n            weighted_variance = (np.sum(left_mask) * left_variance + np.sum(right_mask) * right_variance) / len(y)\n            regularization = self.alpha * np.sum(coefficients ** 2)\n            return weighted_variance + regularization\n\n        initial_params = np.append(np.random.randn(X.shape[1]), 0)\n        result = minimize(objective, initial_params, method=\"BFGS\")\n        self.coefficients, self.bias = result.x[:-1], result.x[-1]\n\n        split = (X @ self.coefficients + self.bias) < 0\n        left_X, right_X = X[split], X[~split]\n        left_y, right_y = y[split], y[~split]\n\n        if len(left_y) == 0 or len(right_y) == 0:\n            self.is_leaf = True\n            self.prediction = np.mean(y)\n            return\n\n        self.left = ObliqueDecisionTreeNode(depth=self.depth + 1, max_depth=self.max_depth, alpha=self.alpha, \n                                             min_samples_split=self.min_samples_split, min_samples_leaf=self.min_samples_leaf)\n        self.left.fit(left_X, left_y)\n        self.right = ObliqueDecisionTreeNode(depth=self.depth + 1, max_depth=self.max_depth, alpha=self.alpha, \n                                              min_samples_split=self.min_samples_split, min_samples_leaf=self.min_samples_leaf)\n        self.right.fit(right_X, right_y)\n\n    def predict(self, X):\n        if self.is_leaf:\n            return self.prediction\n        decision = (X @ self.coefficients + self.bias) < 0\n        if decision:\n            return self.left.predict(X)\n        else:\n            return self.right.predict(X)\n\nclass ObliqueDecisionTreeRegressor(BaseEstimator, RegressorMixin):\n    def __init__(self, max_depth=5, alpha=0.01, n_components=20, min_samples_split=2, min_samples_leaf=1, verbose=0):\n        self.max_depth = max_depth\n        self.alpha = alpha\n        self.n_components = n_components\n        self.min_samples_split = min_samples_split\n        self.min_samples_leaf = min_samples_leaf\n        self.verbose = verbose\n\n        self.root = ObliqueDecisionTreeNode(max_depth=max_depth, alpha=alpha, \n                                             min_samples_split=min_samples_split, min_samples_leaf=min_samples_leaf)\n        self.scaler = StandardScaler()\n        self.pca = PCA(n_components=n_components)\n        self.imputer = SimpleImputer(strategy='median')\n\n    def fit(self, X, y):\n        X_imputed = self.imputer.fit_transform(X)\n\n        if hasattr(y, 'values'):\n            y = y.values\n\n        X_train, X_valid, y_train, y_valid = train_test_split(X_imputed, y, test_size=0.2, random_state=42)\n\n        X_train_scaled = self.scaler.fit_transform(X_train)\n        X_train_reduced = self.pca.fit_transform(X_train_scaled)\n\n        self.root.fit(X_train_reduced, y_train)\n\n        y_valid_scaled = self.scaler.transform(X_valid)\n        # y_valid_reduced = self.pca.transform(y_valid.reshape(-1, 1))  \n\n        y_valid_pred = self.predict(X_valid)  \n        validation_mse = mean_squared_error(y_valid, y_valid_pred)\n        print(\"Validation MSE:\", validation_mse)\n\n        return self\n\n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        X_scaled = self.scaler.transform(X_imputed)\n        X_reduced = self.pca.transform(X_scaled)\n        return np.array([self.root.predict(x) for x in X_reduced])\n\nODT_Params = {\n    'max_depth': 5,\n    'alpha': 0.01,\n    'n_components': 20,\n    'min_samples_split': 2,\n    'min_samples_leaf': 1,\n    'verbose': 1\n}","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.199928Z","iopub.status.idle":"2024-11-28T04:14:23.200361Z","shell.execute_reply.started":"2024-11-28T04:14:23.200136Z","shell.execute_reply":"2024-11-28T04:14:23.20016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'exact'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)\nODT_Model = ObliqueDecisionTreeRegressor(**ODT_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model),\n    #('odt', ODT_Model),\n])","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.201522Z","iopub.status.idle":"2024-11-28T04:14:23.201991Z","shell.execute_reply.started":"2024-11-28T04:14:23.201735Z","shell.execute_reply":"2024-11-28T04:14:23.201775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission1 = TrainML(voting_model, test)\n\n# Save submission\n# Submission1.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.203621Z","iopub.status.idle":"2024-11-28T04:14:23.20395Z","shell.execute_reply.started":"2024-11-28T04:14:23.20378Z","shell.execute_reply":"2024-11-28T04:14:23.203804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission1","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.205181Z","iopub.status.idle":"2024-11-28T04:14:23.205468Z","shell.execute_reply.started":"2024-11-28T04:14:23.205327Z","shell.execute_reply":"2024-11-28T04:14:23.205341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)\nODT_Model = ObliqueDecisionTreeRegressor(**ODT_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    #('tabnet', TabNet_Model),\n    ('odt', ODT_Model),\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.206944Z","iopub.status.idle":"2024-11-28T04:14:23.207246Z","shell.execute_reply.started":"2024-11-28T04:14:23.207104Z","shell.execute_reply":"2024-11-28T04:14:23.20712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission2","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.208397Z","iopub.status.idle":"2024-11-28T04:14:23.208656Z","shell.execute_reply.started":"2024-11-28T04:14:23.208528Z","shell.execute_reply":"2024-11-28T04:14:23.208542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb',    Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb',    Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat',    Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf',     Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb',     Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))])),\n    #('tabnet', Pipeline(steps=[('imputer', imputer), ('regressor', TabNetWrapper(**TabNet_Params))])),\n    #('odt',    Pipeline(steps=[('imputer', imputer), ('regressor', ObliqueDecisionTreeRegressor(**ODT_Params))])),\n])\n\nSubmission3 = TrainML(ensemble, test)","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.210139Z","iopub.status.idle":"2024-11-28T04:14:23.210443Z","shell.execute_reply.started":"2024-11-28T04:14:23.210295Z","shell.execute_reply":"2024-11-28T04:14:23.21031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.211577Z","iopub.status.idle":"2024-11-28T04:14:23.212045Z","shell.execute_reply.started":"2024-11-28T04:14:23.211809Z","shell.execute_reply":"2024-11-28T04:14:23.211834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.213324Z","iopub.status.idle":"2024-11-28T04:14:23.213745Z","shell.execute_reply.started":"2024-11-28T04:14:23.213519Z","shell.execute_reply":"2024-11-28T04:14:23.213541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_submission","metadata":{"execution":{"iopub.status.busy":"2024-11-28T04:14:23.215111Z","iopub.status.idle":"2024-11-28T04:14:23.215545Z","shell.execute_reply.started":"2024-11-28T04:14:23.215307Z","shell.execute_reply":"2024-11-28T04:14:23.215331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}