{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":1003630,"sourceType":"datasetVersion","datasetId":442595},{"sourceId":2660070,"sourceType":"datasetVersion","datasetId":686792},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302},{"sourceId":10161923,"sourceType":"datasetVersion","datasetId":6275040},{"sourceId":10162258,"sourceType":"datasetVersion","datasetId":6275292}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:30:52.331214Z","iopub.execute_input":"2024-12-22T14:30:52.331560Z","iopub.status.idle":"2024-12-22T14:30:56.785561Z","shell.execute_reply.started":"2024-12-22T14:30:52.331525Z","shell.execute_reply":"2024-12-22T14:30:56.784623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --no-index --no-deps /kaggle/input/pytorchtransformer/tab_transformer_pytorch-0.3.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:30:56.786473Z","iopub.execute_input":"2024-12-22T14:30:56.786720Z","iopub.status.idle":"2024-12-22T14:30:57.781497Z","shell.execute_reply.started":"2024-12-22T14:30:56.786700Z","shell.execute_reply":"2024-12-22T14:30:57.780708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --no-index --no-deps /kaggle/input/pytorch/einops-0.8.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:30:57.782565Z","iopub.execute_input":"2024-12-22T14:30:57.782793Z","iopub.status.idle":"2024-12-22T14:30:58.619578Z","shell.execute_reply.started":"2024-12-22T14:30:57.782775Z","shell.execute_reply":"2024-12-22T14:30:58.618722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:30:58.620497Z","iopub.execute_input":"2024-12-22T14:30:58.620729Z","iopub.status.idle":"2024-12-22T14:31:13.001043Z","shell.execute_reply.started":"2024-12-22T14:30:58.620710Z","shell.execute_reply":"2024-12-22T14:31:13.000332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data processing\n","metadata":{}},{"cell_type":"code","source":"## Time-series data statistics extraction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:31:13.001856Z","iopub.execute_input":"2024-12-22T14:31:13.002687Z","iopub.status.idle":"2024-12-22T14:31:13.006150Z","shell.execute_reply.started":"2024-12-22T14:31:13.002650Z","shell.execute_reply":"2024-12-22T14:31:13.005267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def dictionary_of_statistics(data, time = None):\n    \n    #Aggreate statistics for the dictionary\n    stats_summary = data.agg(['mean', 'median', 'max', 'std']).to_dict()\n    \n    flattened_stats = {}\n    for col, stats in stats_summary.items():\n        for stat_name, value in stats.items():\n            key = f\"{stat_name}_{col}\"\n            if (time is not None):\n                key = f\"{stat_name}_{col}_{time}\"\n            flattened_stats[key] = value\n    \n    return flattened_stats\n\n#Feature engineering\n\ndef compute_time_features(data, day_start_hour=6, day_end_hour=18, expected_diff=5):\n    \n    \"\"\"\n    Compute and add time-related features to the DataFrame.\n\n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'time_of_day' in nanosecond and 'relative_date_PCIAT'.\n    - day_start_hour (int): Hour to start the day period. Default is 8.\n    - day_end_hour (int): Hour to end the day period. Default is 21.\n    - expected_diff (int): Expected time difference between steps in seconds. Default is 5.\n    \n    \"\"\"\n    \n    #From nanosecond to hour in a day\n    data['time_of_day_hours'] = data['time_of_day'] / 1e9 / 3600\n    data['day_time'] = data['relative_date_PCIAT'] + data['time_of_day_hours'] / 24\n    \n    #Categorize the day and night based on time data\n    \n    data['day_period'] = np.where(\n        (data['time_of_day_hours'] >= day_start_hour) &\n        (data['time_of_day_hours'] < day_end_hour),\n        'day', 'night'\n    )\n    \n    #Time difference beween steps\n    #As the description, the time_of_day should represent the start of a 5s window over which the data was sampled\n    #Calculate the time difference between each step\n    data['time_diff'] = (data['day_time'].diff() * 86400).round(0) # seconds in a day\n    data['measurement_after_gap'] = data['time_diff'] > expected_diff\n    \ndef no_motion_periods(worn_data):\n    \"\"\"\n    Find periods of no motion and give analytical insights in the data.\n\n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'time_of_day' and 'relative_date_PCIAT'.\n\n    Returns:\n    - pd.DataFrame: DataFrame with new features: \n    + total duration of no motion periods per day\n    + the number of no motion periods per day.\n    \"\"\"\n    \n    #Calculate no motion periods\n    no_motion = worn_data['enmo'] == 0\n    motion_group = (\n        (no_motion != no_motion.shift()) |\n        (worn_data['measurement_after_gap'])\n    ).cumsum()\n\n    no_motion_periods = worn_data[no_motion].groupby(\n        motion_group\n    )['day_time'].agg(['min', 'max'])\n\n    no_motion_periods['duration_sec'] = (\n        (no_motion_periods['max'] - no_motion_periods['min']) * 86400\n    ).round(0).astype(int)\n    \n    no_motion_periods['duration_sec'] += 5\n    no_motion_periods['day'] = no_motion_periods['min'].astype(int)\n    \n    # Calculate daily statistics on no motion periods\n    daily_stats = no_motion_periods.groupby(no_motion_periods['day']) \\\n        .agg(no_motion_duration=('duration_sec', 'sum'),\n            no_motion_count=('duration_sec', 'size'))\n    \n    #Aggreate statistics for the dictionary\n    return dictionary_of_statistics(daily_stats)\n\ndef circadian_rhythm_analysis(worn_data):\n    \n    \"\"\"\n    Make features capturing the variation in activity across the 24-hour cycle, \n    separately for day and night times (or wakefulness and sleep periods).\n    \n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'time_of_day_hours', 'day_period' and 'relative_date_PCIAT'.\n    \n    Returns:\n    - pd.DataFrame: 2 DataFrame, corresponding to day and night times,  with new features capturing the circadian rhythm of the wearer,\n    + Standard deviation across hourly means per day\n    + Peak hour of activity per day\n    + Entropy of activity distribution per day\n    \"\"\"\n    \n    hourly_activity = worn_data.groupby(\n        [worn_data['relative_date_PCIAT'].astype(int),\n        worn_data['time_of_day_hours'].astype(int),\n        worn_data['day_period']]\n    )['enmo'].agg(['mean', 'max'])\n\n    features = hourly_activity['mean'].groupby(\n        ['relative_date_PCIAT', 'day_period']\n    ).agg(\n        std_across_hours='std',\n        peak_hour=lambda x: x.idxmax()[1],\n        entropy=lambda x: -(x / x.sum() * np.log(x / x.sum() + 1e-9)).sum()\n    )\n    \n    day_features = features.xs('day', level='day_period')\n    night_features = features.xs('night', level='day_period')\n    \n    return dictionary_of_statistics(day_features, time=\"day\") | dictionary_of_statistics(night_features, time='night')\n    # return features\n\ndef physical_activity_analysis(worn_data):\n        \n    \"\"\"\n    Analyze the Moderate to Vigorous Physical Activity (MVPA) based on a threshold of ENMO values, \n    and calculate the duration of the detected MVPA activity bouts\n    \n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'enmo', 'time_diff' and 'day_time'.\n    \n    Returns:\n    - pd.DataFrame: DataFrame with new features capturing the physical activity level of the wearer,\n    including:\n    + Total duration of MVPA per day\n    + Number of MVPA periods per day\n    \"\"\"\n    # In order to classify physical activity as MVPA, we only retained activities that lasted at least 1 minute and met the criteria for the 100 mg (= 0.1g) threshold\n    mvpa_threshold = 0.1\n    merge_gap = 60\n    \n    def merge_mvpa_groups(df, allowed_gap=60, merge_gap=60):\n        last_mvpa_time = df['day_time'].where(df['is_mvpa']).ffill().shift()\n        \n        mvpa_time_diff = (\n            (df['day_time'] - last_mvpa_time) * 86400\n        ).round(0)\n        \n        mvpa_group = (\n            (df['is_mvpa'] != df['is_mvpa'].shift()) |\n            (df['time_diff'] >= allowed_gap)\n        ).cumsum()\n        \n        is_mvpa_start = (\n            (mvpa_group != mvpa_group.shift()) &\n            df['is_mvpa']\n        )\n        \n        group_increment = is_mvpa_start & (\n            (mvpa_time_diff >= merge_gap) | last_mvpa_time.isnull()\n        )\n        \n        merged_group = group_increment.cumsum()\n        merged_group.loc[~df['is_mvpa']] = np.nan\n        \n        return merged_group\n    \n    worn_data['is_mvpa'] = worn_data['enmo'] > mvpa_threshold\n    worn_data['mvpa_merged_group'] = merge_mvpa_groups(worn_data)\n\n    mvpa_periods = worn_data[\n        worn_data['is_mvpa']\n    ].groupby('mvpa_merged_group')['day_time'].agg(['min', 'max'])\n\n    mvpa_periods['duration_sec'] = (\n        mvpa_periods['max'] - mvpa_periods['min']\n    ) * 86400  # days to seconds\n\n    mvpa_periods = mvpa_periods[mvpa_periods['duration_sec'] >= 60]\n    mvpa_periods['duration_min'] = mvpa_periods['duration_sec'] / 60\n    \n    mvpa_periods['day'] = mvpa_periods['min'].astype(int)\n\n    daily_stats = mvpa_periods.groupby(mvpa_periods['day']) \\\n        .agg(mvpa_total_duration=('duration_sec', 'sum'),\n            mvpa_count_periods=('duration_sec', 'size'))\n        \n    return dictionary_of_statistics(daily_stats)\n    \ndef activity_transition_analysis(worn_data):\n        \n    \"\"\"\n    The analysis to look at transitions between low, moderate and vigorous activity. To smooth out sudden, \n    short bursts of different activities, we filter out segments with a duration below a 1 minute threshold.\n    \n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'enmo', 'time_diff' and 'day_time'.\n    \n    Returns:\n    - pd.DataFrame: DataFrame with new features capturing the activity transitions of the wearer,\n    + Total duration of different of activity per day\n    + Number of different activity periods per day\n    \"\"\"\n    mvpa_threshold = 0.1\n    vig_threshold = 0.5\n    worn_data['activity_type'] = pd.cut(\n        worn_data['enmo'],\n        bins=[-np.inf, mvpa_threshold, vig_threshold, np.inf],\n        labels=['low', 'moderate', 'vigorous']\n    )\n    activity_group = (\n        (worn_data['activity_type'] != worn_data['activity_type'].shift()) |\n        (worn_data['measurement_after_gap'])\n    ).cumsum()\n    \n    activity_periods = worn_data.groupby(activity_group).agg(\n        min=('day_time', 'min'),\n        max=('day_time', 'max'),\n        activity_type=('activity_type', 'first')\n    )\n    activity_periods['duration_sec'] = (\n        activity_periods['max'] - activity_periods['min']\n    ) * 86400 + 5 \n\n    activity_periods = activity_periods[activity_periods['duration_sec'] >= 60]\n    activity_periods['duration_min'] = activity_periods['duration_sec'] / 60\n    \n    activity_periods['day'] = activity_periods['min'].astype(int)\n    activity_periods['transition_num'] = (\n        activity_periods.groupby('day')['activity_type']\n        .apply(lambda x: (x != x.shift()).cumsum())\n        .reset_index(level=0, drop=True)\n    )\n    \n    low_activity = activity_periods[activity_periods['activity_type'] == 'low'].groupby('day').agg(\n        low_act_total_duration=('duration_sec', 'sum'),\n        low_act_count_periods=('duration_sec', 'size')\n    )\n    \n    moderate_activity = activity_periods[activity_periods['activity_type'] == 'moderate'].groupby('day').agg(\n            moderate_act_total_duration=('duration_sec', 'sum'),\n            moderate_act_count_periods=('duration_sec', 'size')\n    )\n\n        \n    daily_transitions = activity_periods.groupby('day').agg(\n        count_transitions=('transition_num', 'max')\n    )\n    \n    return dictionary_of_statistics(low_activity) | dictionary_of_statistics(moderate_activity) | dictionary_of_statistics(daily_transitions)\n\ndef activity_light_exposure(worn_data):\n        \n    \"\"\"\n    Analyze the correlation between light exposure and physical activity level.\n    \n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'light' and 'enmo'.\n    \n    Returns:\n    - float: The correlation between light exposure and physical activity level.\n    \"\"\"\n    correlation_light_enmo = worn_data[['light', 'enmo']].corr().iloc[0, 1]\n    return {'correlation_light_enmo': correlation_light_enmo}\n\ndef process_file(file_path, participant_id):\n    data = pd.read_parquet(file_path)\n\n    # Compute time features\n    compute_time_features(data)\n        \n    # Calculate the percentage of non-worn time\n    non_wear_percentage = (data['non-wear_flag'].sum() / len(data)) * 100\n    \n    # Filter out the worn data\n    worn_data = data[data['non-wear_flag'] == 0]\n    \n    # recalculate time difference between rows and measurement_after_gap flag in the worn data\n    expected_diff = 5\n    worn_data['time_diff'] = (worn_data['day_time'].diff() * 86400).round(0)\n    worn_data['measurement_after_gap'] = worn_data['time_diff'] > expected_diff\n\n    # Compute no motion periods\n    no_motion_stats = no_motion_periods(worn_data)\n    # Circadian rhythm analysis\n    circadian_rhythm_stats = circadian_rhythm_analysis(worn_data)\n    # Physical activity analysis\n    physical_activity_stats = physical_activity_analysis(worn_data)\n    # Activity transition analysis\n    activity_transition_stats = activity_transition_analysis(worn_data)\n    # Compute activity light correlation\n    activity_light_stats = activity_light_exposure(worn_data)\n\n    return {\n        'id': participant_id,\n    } | no_motion_stats | circadian_rhythm_stats | physical_activity_stats | activity_transition_stats | activity_light_stats\n\ndef load_time_series(dir_name):\n    \n    participant_ids = os.listdir(dir_name)\n    \n    with ThreadPoolExecutor() as executor:\n        #tqdm: Wraps the executor.map iterable with tqdm -> Show a progress bar indicating the processing status\n        results = list(tqdm(executor.map(lambda x: process_file(os.path.join(dir_name, x, 'part-0.parquet'), x), participant_ids), total=len(participant_ids)))\n\n    df = pd.DataFrame(results)\n    # Replace inf and -inf with NaN\n    df.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n    # Replace NaN with 0\n    df.fillna(0, inplace=True)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:31:13.008784Z","iopub.execute_input":"2024-12-22T14:31:13.009143Z","iopub.status.idle":"2024-12-22T14:31:13.031625Z","shell.execute_reply.started":"2024-12-22T14:31:13.009108Z","shell.execute_reply":"2024-12-22T14:31:13.030966Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Build a AutoEncoder to compress size of time-series dataset, helps to remove redundant features and avoid overfitting**","metadata":{}},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:31:13.033156Z","iopub.execute_input":"2024-12-22T14:31:13.033376Z","iopub.status.idle":"2024-12-22T14:31:13.050151Z","shell.execute_reply.started":"2024-12-22T14:31:13.033338Z","shell.execute_reply":"2024-12-22T14:31:13.049574Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"markdown","source":"**Loading tabular data**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:31:13.051013Z","iopub.execute_input":"2024-12-22T14:31:13.051318Z","iopub.status.idle":"2024-12-22T14:31:13.138169Z","shell.execute_reply.started":"2024-12-22T14:31:13.051296Z","shell.execute_reply":"2024-12-22T14:31:13.137239Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Few SII scores are still derived from the sum of NAN values in PICAT questions, leading to potentially invalid SII values. The below code tries to estimate the severity of internet usage SII based on current SII and the maximum possible SII**","metadata":{}},{"cell_type":"code","source":"#Generate a list of column names in the format PCIAT-PCIAT_XX\nPCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\n\n#Recalculates the SII value based on the the current PCIAT values and the possible maximum PCIAT values\ndef recalculate_sii(row):\n    value = 0\n    if (not pd.isna(row['PCIAT-PCIAT_Total'])):\n        value = row['PCIAT-PCIAT_Total']\n        \n    max_possible = value + row[PCIAT_cols].isna().sum() * 5\n    \n    if value <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= value <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= value <= 79 and max_possible <= 79:\n        return 2\n    elif value >= 80 and max_possible >= 80:\n        return 3\n    \n    return np.nan\n\ntrain['recalc_sii'] = train.apply(recalculate_sii, axis=1)\ntrain['sii'] = train['recalc_sii']\ntrain.drop(columns='recalc_sii', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:31:13.139040Z","iopub.execute_input":"2024-12-22T14:31:13.139320Z","iopub.status.idle":"2024-12-22T14:31:14.694880Z","shell.execute_reply.started":"2024-12-22T14:31:13.139294Z","shell.execute_reply":"2024-12-22T14:31:14.694132Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Loading time-series data**","metadata":{}},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:31:14.695737Z","iopub.execute_input":"2024-12-22T14:31:14.696035Z","iopub.status.idle":"2024-12-22T14:35:05.105285Z","shell.execute_reply.started":"2024-12-22T14:31:14.696007Z","shell.execute_reply":"2024-12-22T14:35:05.104548Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Perform auto encoding on the time_series data**","metadata":{}},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:05.106225Z","iopub.execute_input":"2024-12-22T14:35:05.106499Z","iopub.status.idle":"2024-12-22T14:35:14.613958Z","shell.execute_reply.started":"2024-12-22T14:35:05.106468Z","shell.execute_reply":"2024-12-22T14:35:14.613061Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Take a look at the time-series data before and after performing auto encoding. There is a minimal difference in data dimensionality, which means that the feature engineering approach applied to this dataset is effective**","metadata":{}},{"cell_type":"code","source":"print(df_train.shape, train_ts_encoded.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:14.614953Z","iopub.execute_input":"2024-12-22T14:35:14.615751Z","iopub.status.idle":"2024-12-22T14:35:14.620466Z","shell.execute_reply.started":"2024-12-22T14:35:14.615726Z","shell.execute_reply":"2024-12-22T14:35:14.619593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain_ts_encoded['id'] = train_ts_encoded['id'].str.replace('id=', '')\ntest_ts_encoded['id'] = test_ts_encoded['id'].str.replace('id=', '')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:14.621244Z","iopub.execute_input":"2024-12-22T14:35:14.621483Z","iopub.status.idle":"2024-12-22T14:35:14.639827Z","shell.execute_reply.started":"2024-12-22T14:35:14.621461Z","shell.execute_reply":"2024-12-22T14:35:14.639090Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data merging","metadata":{}},{"cell_type":"code","source":"train = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:14.640540Z","iopub.execute_input":"2024-12-22T14:35:14.640739Z","iopub.status.idle":"2024-12-22T14:35:14.671299Z","shell.execute_reply.started":"2024-12-22T14:35:14.640722Z","shell.execute_reply":"2024-12-22T14:35:14.670555Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Take a look at the test set after grafting encoded time-series data. As shown below, the participants who didn't wear device have features related to time-series values are NaN. To handle, we will replace all NaN values with 0**","metadata":{}},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:14.672193Z","iopub.execute_input":"2024-12-22T14:35:14.672455Z","iopub.status.idle":"2024-12-22T14:35:14.752748Z","shell.execute_reply.started":"2024-12-22T14:35:14.672434Z","shell.execute_reply":"2024-12-22T14:35:14.751908Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handle missing in predictors","metadata":{}},{"cell_type":"code","source":"train_cols = set(train.columns)\ntest_cols = set(test.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\ncolumns_not_in_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:14.753429Z","iopub.execute_input":"2024-12-22T14:35:14.753705Z","iopub.status.idle":"2024-12-22T14:35:14.759146Z","shell.execute_reply.started":"2024-12-22T14:35:14.753672Z","shell.execute_reply":"2024-12-22T14:35:14.758432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Initialize a KNNImputer for handling missing valuee. Statistically, KNNImputer outweights SimpleImputer in this context. However, it encounters the computationally expensive cost, and the requirement of hyperparameter K (K value should be considered to change later)**","metadata":{}},{"cell_type":"code","source":"# Get numeric columns excluding 'sii'\nnumeric_cols = train.select_dtypes(include=['float64', 'int64', 'int32', 'float32']).columns.tolist()\nnumeric_cols.remove('sii')\n\n# Handle numerical with KNNImputer\nimputer = KNNImputer(n_neighbors=5)\ntrain[numeric_cols] = imputer.fit_transform(train[numeric_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:14.759987Z","iopub.execute_input":"2024-12-22T14:35:14.760268Z","iopub.status.idle":"2024-12-22T14:35:27.412994Z","shell.execute_reply.started":"2024-12-22T14:35:14.760240Z","shell.execute_reply":"2024-12-22T14:35:27.412302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Get categorical columns\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\n#File the nan values in categorical collumns with 'Missing'\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\n#Convert from categorical to numeric \ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.413862Z","iopub.execute_input":"2024-12-22T14:35:27.414185Z","iopub.status.idle":"2024-12-22T14:35:27.466034Z","shell.execute_reply.started":"2024-12-22T14:35:27.414143Z","shell.execute_reply":"2024-12-22T14:35:27.465447Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FEATURE ENGINEERING","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n\n    # Replace any remaining inf values with NaN\n    df = df.replace([np.inf, -np.inf], np.nan)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.469338Z","iopub.execute_input":"2024-12-22T14:35:27.469598Z","iopub.status.idle":"2024-12-22T14:35:27.475044Z","shell.execute_reply.started":"2024-12-22T14:35:27.469577Z","shell.execute_reply":"2024-12-22T14:35:27.474230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = feature_engineering(train)\n#Remove rows which has got fewer than 10 non-null values\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.476637Z","iopub.execute_input":"2024-12-22T14:35:27.476914Z","iopub.status.idle":"2024-12-22T14:35:27.534791Z","shell.execute_reply.started":"2024-12-22T14:35:27.476883Z","shell.execute_reply":"2024-12-22T14:35:27.534140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii',\n                'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW','BMI_PHR']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\n\nfeaturesCols.remove('sii')\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.535536Z","iopub.execute_input":"2024-12-22T14:35:27.535784Z","iopub.status.idle":"2024-12-22T14:35:27.542930Z","shell.execute_reply.started":"2024-12-22T14:35:27.535765Z","shell.execute_reply":"2024-12-22T14:35:27.542090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Check and replace any INF values with NaN\nif np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.543555Z","iopub.execute_input":"2024-12-22T14:35:27.543836Z","iopub.status.idle":"2024-12-22T14:35:27.560012Z","shell.execute_reply.started":"2024-12-22T14:35:27.543814Z","shell.execute_reply":"2024-12-22T14:35:27.559402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = train\n# Get columns with NaN values\nnan_columns = df.columns[df.isna().any()].tolist()\n\n# Calculate number of NaN values per column\nnan_counts = df[nan_columns].isna().sum()\n\n# Sort by number of NaN values (descending)\nnan_counts_sorted = nan_counts.sort_values(ascending=False)\n\n# Display results\nprint(\"\\nColumns with NaN values:\")\nprint(\"-\" * 50)\nfor col, count in nan_counts_sorted.items():\n    total = len(df)\n    percentage = (count/total * 100)\n    print(f\"{col:<30} {count:>7} NaN values ({percentage:>6.2f}%)\")\n\nprint(f\"\\nTotal columns with NaN values: {len(nan_columns)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.560874Z","iopub.execute_input":"2024-12-22T14:35:27.561098Z","iopub.status.idle":"2024-12-22T14:35:27.577675Z","shell.execute_reply.started":"2024-12-22T14:35:27.561069Z","shell.execute_reply":"2024-12-22T14:35:27.577039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.dropna(subset=['BMR_Weight', 'DEE_Weight', 'Hydration_Status'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.578478Z","iopub.execute_input":"2024-12-22T14:35:27.578694Z","iopub.status.idle":"2024-12-22T14:35:27.593136Z","shell.execute_reply.started":"2024-12-22T14:35:27.578676Z","shell.execute_reply":"2024-12-22T14:35:27.592400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train.to_csv(\"train1.csv\", index=False)\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.593857Z","iopub.execute_input":"2024-12-22T14:35:27.594104Z","iopub.status.idle":"2024-12-22T14:35:27.607045Z","shell.execute_reply.started":"2024-12-22T14:35:27.594075Z","shell.execute_reply":"2024-12-22T14:35:27.606436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# VIME Definition\n## Mask & Pretext generator\n**The below cell defines severals utilities for building VIME, including mask_generator & pretext_generator**","metadata":{}},{"cell_type":"code","source":"def mask_generator (p_m, x):\n  \"\"\"Generate mask vector.\n  \n  Args:\n    - p_m: corruption probability\n    - x: feature matrix\n    \n  Returns:\n    - mask: binary mask matrix \n  \"\"\"\n  mask = np.random.binomial(1, p_m, x.shape)\n  return mask\n\ndef pretext_generator (m, x):  \n  \"\"\"Generate corrupted samples.\n  \n  Args:\n    m: mask matrix\n    x: feature matrix\n    \n  Returns:\n    m_new: final mask matrix after corruption\n    x_tilde: corrupted feature matrix\n  \"\"\"\n  \n  # Parameters\n  no, dim = x.shape  \n  # Randomly (and column-wise) shuffle data\n  x_bar = np.zeros([no, dim])\n  for i in range(dim):\n    idx = np.random.permutation(no)\n    x_bar[:, i] = x[idx, i]\n    \n  # Corrupt samples\n  x_tilde = x * (1-m) + x_bar * m  \n  # Define new mask matrix\n  m_new = 1 * (x != x_tilde)\n\n  return m_new, x_tilde\n\ndef convert_matrix_to_vector(matrix):\n  \"\"\"Convert two dimensional matrix into one dimensional vector\n  \n  Args:\n    - matrix: two dimensional matrix\n    \n  Returns:\n    - vector: one dimensional vector\n  \"\"\"\n  # Parameters\n  no, dim = matrix.shape\n  # Define output  \n  vector = np.zeros([no,])\n  \n  # Convert matrix to vector\n  for i in range(dim):\n    idx = np.where(matrix[:, i] == 1)\n    vector[idx] = i\n    \n  return vector\n\ndef convert_vector_to_matrix(vector):\n  \"\"\"Convert one dimensional vector into two dimensional matrix\n  \n  Args:\n    - vector: one dimensional vector\n    \n  Returns:\n    - matrix: two dimensional matrix\n  \"\"\"\n  # Parameters\n  no = len(vector)\n  dim = len(np.unique(vector))\n  # Define output\n  matrix = np.zeros([no,dim])\n  \n  # Convert vector to matrix\n  for i in range(dim):\n    idx = np.where(vector == i)\n    matrix[idx, i] = 1\n    \n  return matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.607692Z","iopub.execute_input":"2024-12-22T14:35:27.607870Z","iopub.status.idle":"2024-12-22T14:35:27.618641Z","shell.execute_reply.started":"2024-12-22T14:35:27.607855Z","shell.execute_reply":"2024-12-22T14:35:27.617947Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## VIME Self-supervised framework","metadata":{}},{"cell_type":"code","source":"# Necessary packages\nfrom tensorflow import keras\nfrom tensorflow.keras import Model, Input, layers\n\ndef vime_self (x_unlab, p_m, alpha, parameters):\n  \"\"\"Self-supervised learning part in VIME.\n  \n  Args:\n    x_unlab: unlabeled feature\n    p_m: corruption probability\n    alpha: hyper-parameter to control the weights of feature and mask losses\n    parameters: epochs, batch_size\n    \n  Returns:\n    encoder: Representation learning block\n  \"\"\"\n    \n  # Parameters\n  _, dim = x_unlab.shape\n  epochs = parameters['epochs']\n  batch_size = parameters['batch_size']\n  \n  # Build model using Functional API\n  inputs = Input(shape=(dim,))\n  # Encoder\n  h = layers.Dense(dim, activation='relu')(inputs)\n  # Mask estimator\n  mask_output = layers.Dense(dim, activation='sigmoid', name='mask')(h)\n  # Feature estimator\n  feature_output = layers.Dense(dim, activation='sigmoid', name='feature')(h)\n  \n  #Create model\n  model = Model(inputs = inputs, outputs = [mask_output, feature_output])\n  \n  model.compile(optimizer='rmsprop',\n                loss={'mask': 'binary_crossentropy', \n                      'feature': 'mean_squared_error'},\n                loss_weights={'mask':1.0\n                              , 'feature':float(alpha)})\n  \n  # Generate corrupted samples\n  m_unlab = mask_generator(p_m, x_unlab)\n  m_label, x_tilde = pretext_generator(m_unlab, x_unlab)\n  \n  # Fit model on unlabeled data\n  model.fit(x_tilde, {'mask': m_label, 'feature': x_unlab}, \n            epochs = epochs, batch_size= batch_size)\n      \n  # Extract encoder\n  encoder = Model(\n      inputs=model.input,\n      outputs=model.layers[1].output,\n      name='encoder'\n  )\n  \n  return encoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.619439Z","iopub.execute_input":"2024-12-22T14:35:27.619644Z","iopub.status.idle":"2024-12-22T14:35:27.659248Z","shell.execute_reply.started":"2024-12-22T14:35:27.619627Z","shell.execute_reply":"2024-12-22T14:35:27.658664Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## VIME Semi-supervised framework","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nimport numpy as np\nfrom tensorflow import keras\n \n\"\"\"\nExpected flow:\n- Load & process data\n- Create network architecture\n- Train the model using:\n+ Supervised loss from labeled data\n+ Unsupervised loss from unlabeled data\n\"\"\"\n\nclass Predictor(keras.Model):\n    def __init__(self, hidden_dim, label_dim):\n        super(Predictor, self).__init__()\n        #Define layers\n        self.dense1 = keras.layers.Dense(hidden_dim, activation='relu')\n        self.dense2 = keras.layers.Dense(hidden_dim, activation='relu')\n        self.output_layer = keras.layers.Dense(label_dim)\n        \n    def call(self, x_input):\n        #Forward pass\n        inter_layer = self.dense1(x_input)\n        inter_layer = self.dense2(inter_layer)\n        y_hat_logit = self.output_layer(inter_layer)\n        y_hat = tf.nn.softmax(y_hat_logit)\n        return y_hat_logit, y_hat\n \ndef vime_semi(x_train, y_train, x_unlab, x_test, parameters, \n              p_m, K, beta, file_name):\n    \"\"\"Semi-supervied learning part in VIME.\n    \n    Args:\n        - x_train, y_train: training dataset\n        - x_unlab: unlabeled dataset\n        - x_test: testing features\n        - parameters: network parameters (hidden_dim, batch_size, iterations)\n        - p_m: corruption probability\n        - K: number of augmented samples\n        - beta: hyperparameter to control supervised and unsupervised loss\n        - file_name: saved filed name for the encoder function\n        \n    Returns:\n        - y_test_hat: prediction on x_test\n    \"\"\"\n    class WeightedKappaLoss(tf.keras.losses.Loss):\n        def __init__(self, num_classes, name=\"weighted_kappa_loss\"):\n            super().__init__(name=name)\n            self.num_classes = num_classes\n            # Create weight matrix\n            weights = tf.cast(tf.range(num_classes), tf.float32)\n            weights = tf.expand_dims(weights, 0) - tf.expand_dims(weights, 1)\n            self.weights = tf.square(weights)\n            \n        @tf.function\n        def call(self, y_true, y_pred):\n            # Convert types\n            y_true = tf.cast(y_true, tf.float32)\n            y_pred = tf.cast(y_pred, tf.float32)\n            \n            # Apply softmax\n            y_pred = tf.nn.softmax(y_pred)\n            \n            # Calculate confusion matrix\n            batch_size = tf.cast(tf.shape(y_true)[0], tf.float32)\n            confusion = tf.matmul(y_true, y_pred, transpose_a=True)\n            confusion = confusion / batch_size\n            \n            # Calculate agreements\n            observed = tf.reduce_sum(confusion * (1.0 - self.weights))\n            expected_rows = tf.reduce_sum(confusion, axis=1)\n            expected_cols = tf.reduce_sum(confusion, axis=0)\n            expected = tf.reduce_sum(\n                tf.matmul(tf.expand_dims(expected_rows, 1),\n                         tf.expand_dims(expected_cols, 0)) * (1.0 - self.weights)\n            )\n            \n            # Calculate kappa\n            kappa = (observed - expected) / (1.0 - expected + 1e-8)\n            return 1.0 - kappa\n    \n    # Network parameters\n    hidden_dim = parameters['hidden_dim']\n    batch_size = parameters['batch_size']\n    iterations = parameters['iterations']\n      \n    # Basic parameters\n    data_dim = x_train.shape[1]\n    label_dim = y_train.shape[1]\n\n    # Divide training and validation sets (9:1)\n    idx = np.random.permutation(len(x_train))\n    train_idx = idx[:int(len(idx)*0.9)]\n    valid_idx = idx[int(len(idx)*0.9):]\n    \n    x_valid, y_valid = x_train[valid_idx], y_train[valid_idx]\n    x_train, y_train = x_train[train_idx], y_train[train_idx]\n    \n    # Load encoder from self-supervised model\n    encoder = keras.models.load_model(file_name)\n    \n    # Create predictor model\n    predictor = Predictor(hidden_dim, label_dim)\n    optimizer = keras.optimizers.Adam()\n    \n    # Encode validation and testing features\n    x_valid = encoder.predict(x_valid)\n    x_test = encoder.predict(x_test)\n    \n    # Setup checkpointing\n    checkpoint = tf.train.Checkpoint(model=predictor)\n    manager = tf.train.CheckpointManager(\n        checkpoint, './save_model', max_to_keep=1)\n    \n    # Early stopping variables\n    best_valid_loss = float('inf')\n    patience = -1\n    \n    # Initialize loss\n    loss_fn = WeightedKappaLoss(num_classes=label_dim)\n    \n    @tf.function\n    def train_step(x_batch, y_batch, xu_batch):\n        with tf.GradientTape() as tape:\n            #Forward pass\n            y_logits, _ = predictor(x_batch)\n            yv_logits, _ = predictor(xu_batch)\n            \n            # Calculate supervised loss\n            supervised_loss = loss_fn(y_batch, y_logits)\n\n            # Unsupervised loss using variance\n            unsupervised_loss = tf.reduce_mean(\n                tf.math.reduce_variance(yv_logits, axis=0))\n            \n            total_loss = supervised_loss + beta * unsupervised_loss\n        \n        # Compute gradients and update weights\n        gradients = tape.gradient(total_loss, predictor.trainable_variables)\n        optimizer.apply_gradients(zip(gradients, predictor.trainable_variables))\n        \n        return total_loss\n        \n    for it in range(iterations):\n        # Select a batch of labeled data\n        batch_idx = np.random.permutation(len(x_train))[:batch_size]\n        x_batch = x_train[batch_idx]\n        y_batch = y_train[batch_idx]\n        \n        # Encode labeled data\n        x_batch = encoder.predict(x_batch)\n        \n        # Select and augment unlabeled data\n        batch_u_idx = np.random.permutation(len(x_unlab))[:batch_size]\n        xu_batch_ori = x_unlab[batch_u_idx]\n        \n        xu_batch = []\n        for _ in range(K):\n            # Mask vector generation\n            m_batch = mask_generator(p_m, xu_batch_ori)\n            # Pretext generator\n            _, xu_batch_temp = pretext_generator(m_batch, xu_batch_ori)\n            \n            # Encode corrupted samples\n            xu_batch_temp = encoder.predict(xu_batch_temp)\n            xu_batch.append(xu_batch_temp)\n        \n        #Convert list to matrix\n        xu_batch = tf.convert_to_tensor(np.array(xu_batch))\n        \n        # Training step\n        loss = train_step(x_batch, y_batch, xu_batch)\n        \n        # Validation step\n        val_logits, _ = predictor(x_valid)\n        \n        val_loss = loss_fn(y_valid, val_logits)\n        \n        if it % 100 == 0:\n            print(f'Iteration: {it}/{iterations}, Loss: {loss:.4f}, Val Loss: {val_loss:.4f}')\n            \n        # Early stopping\n        if val_loss < best_valid_loss:\n            best_valid_loss = val_loss\n            manager.save()\n            patience = 0\n        else:\n            patience += 1\n            if patience >= 100:\n                break\n            \n        \n    # Restore best model\n    checkpoint.restore(manager.latest_checkpoint)\n    \n    # Generate predictions\n    _, y_test_hat = predictor(x_test)\n    \n    return y_test_hat.numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.660017Z","iopub.execute_input":"2024-12-22T14:35:27.660294Z","iopub.status.idle":"2024-12-22T14:35:27.676137Z","shell.execute_reply.started":"2024-12-22T14:35:27.660266Z","shell.execute_reply":"2024-12-22T14:35:27.675325Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Labeling","metadata":{}},{"cell_type":"markdown","source":"## Split training to label & unlabel","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\ndef load_data(train):\n    df = train.copy()\n    \n    cols = df.columns\n    categorial_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\n    normalized_cols = [col for col in cols if col not in categorial_cols]\n    normalized_cols.remove('sii')\n    \n    scaler = MinMaxScaler()\n    df[normalized_cols] = scaler.fit_transform(df[normalized_cols])\n    \n    # Split into labeled and unlabeled sets\n    labeled_df = df[df['sii'].notna()]\n    unlabeled_df = df[df['sii'].isna()]\n    \n    x_label = labeled_df.drop(columns=['sii'])\n    y_label = labeled_df['sii']\n    y_label = np.asarray(pd.get_dummies(y_label))\n    x_unlabel = unlabeled_df.drop(columns=['sii'])\n    \n    return x_label, y_label, x_unlabel, normalized_cols, scaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.676919Z","iopub.execute_input":"2024-12-22T14:35:27.677106Z","iopub.status.idle":"2024-12-22T14:35:27.694154Z","shell.execute_reply.started":"2024-12-22T14:35:27.677090Z","shell.execute_reply":"2024-12-22T14:35:27.693514Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Load labeled & unlabeled data, and convert them to numpy.narray type**","metadata":{}},{"cell_type":"code","source":"x_label, y_label, x_unlabel, normalized_cols, scaler = load_data(train)\n\nx_label = x_label.to_numpy().astype(np.float32)\nx_unlabel = x_unlabel.to_numpy().astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.694871Z","iopub.execute_input":"2024-12-22T14:35:27.695080Z","iopub.status.idle":"2024-12-22T14:35:27.743805Z","shell.execute_reply.started":"2024-12-22T14:35:27.695062Z","shell.execute_reply":"2024-12-22T14:35:27.743183Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Apply min-max scaler on test set**","metadata":{}},{"cell_type":"code","source":"test[normalized_cols] = scaler.transform(test[normalized_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.744639Z","iopub.execute_input":"2024-12-22T14:35:27.744944Z","iopub.status.idle":"2024-12-22T14:35:27.764864Z","shell.execute_reply.started":"2024-12-22T14:35:27.744913Z","shell.execute_reply":"2024-12-22T14:35:27.764157Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## VIME Usage","metadata":{}},{"cell_type":"markdown","source":"**Define hyperparameters for VIME**","metadata":{}},{"cell_type":"code","source":"#Define hyperparameters\np_m = 0.3 #Corruption probability for self-supervised learning\nalpha = 2.0 #Control the weights of feature and mask losses\nK = 3 #number of augmented samples\nbeta = 1.0 #Control the weights of the supervised and unsupervised losses","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.765653Z","iopub.execute_input":"2024-12-22T14:35:27.765966Z","iopub.status.idle":"2024-12-22T14:35:27.770277Z","shell.execute_reply.started":"2024-12-22T14:35:27.765936Z","shell.execute_reply":"2024-12-22T14:35:27.769644Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train VIME-Self**","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_rows', 20)  # Show only 20 rows\npd.set_option('display.min_rows', 10)  # Minimum rows to show\n# For Jupyter notebook display\nfrom IPython.display import display_html\n\nvime_self_parameters = dict()\nvime_self_parameters['batch_size'] = 128\nvime_self_parameters['epochs'] = 50\nvime_self_encoder = vime_self(x_unlabel, p_m, alpha, vime_self_parameters)\n\n#Save encoder\nif not os.path.exists('save_model'):\n  os.makedirs('save_model')\n\nfile_name = './save_model/encoder_model.h5'\n  \nvime_self_encoder.save(file_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:27.771264Z","iopub.execute_input":"2024-12-22T14:35:27.771557Z","iopub.status.idle":"2024-12-22T14:35:31.981812Z","shell.execute_reply.started":"2024-12-22T14:35:27.771529Z","shell.execute_reply":"2024-12-22T14:35:31.981159Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train VIME_Semi using encoder from VIME_Self**","metadata":{}},{"cell_type":"code","source":"# Train VIME-Semi\nvime_semi_parameters = dict()\nvime_semi_parameters['hidden_dim'] = 50\nvime_semi_parameters['batch_size'] = 128\nvime_semi_parameters['iterations'] = 1000\n\nfile_name = './save_model/encoder_model.h5'\n\ny_test_hat = vime_semi(x_label, y_label, x_unlabel, x_unlabel, \n                       vime_semi_parameters, p_m, K, beta, file_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T14:35:31.982611Z","iopub.execute_input":"2024-12-22T14:35:31.982834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_hat = np.argmax(y_test_hat, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:17:10.832085Z","iopub.execute_input":"2024-12-22T15:17:10.832427Z","iopub.status.idle":"2024-12-22T15:17:10.860938Z","shell.execute_reply.started":"2024-12-22T15:17:10.832401Z","shell.execute_reply":"2024-12-22T15:17:10.859538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count unique values in y_test_hat\nunique_values, value_counts = np.unique(y_test_hat, return_counts=True)\n\n# Display results\nfor value, count in zip(unique_values, value_counts):\n    print(f\"Value {value}: {count} occurrences\")\n\n# Optional: verify total\nprint(f\"\\nTotal elements: {len(y_test_hat)}\")\nprint(f\"Sum of counts: {np.sum(value_counts)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:17:31.887950Z","iopub.execute_input":"2024-12-22T15:17:31.888252Z","iopub.status.idle":"2024-12-22T15:17:31.894798Z","shell.execute_reply.started":"2024-12-22T15:17:31.888228Z","shell.execute_reply":"2024-12-22T15:17:31.893997Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data consolidation","metadata":{}},{"cell_type":"code","source":"# x_train = np.concatenate([x_label, x_unlabel], axis=0)\n# print(x_train.shape)\n# x_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:17:34.338478Z","iopub.execute_input":"2024-12-22T15:17:34.338795Z","iopub.status.idle":"2024-12-22T15:17:34.342134Z","shell.execute_reply.started":"2024-12-22T15:17:34.338768Z","shell.execute_reply":"2024-12-22T15:17:34.341255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# y_label_converted = np.argmax(y_label, axis=1).reshape(-1, 1)\n# y_train = np.concatenate([y_label_converted, y_test_hat.reshape(-1, 1)],axis=0)\n# print(y_train.shape)\n# y_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:17:34.822502Z","iopub.execute_input":"2024-12-22T15:17:34.822783Z","iopub.status.idle":"2024-12-22T15:17:34.826293Z","shell.execute_reply.started":"2024-12-22T15:17:34.822762Z","shell.execute_reply":"2024-12-22T15:17:34.825407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\n# Giả sử các mảng sau đây đã được định nghĩa\n# x_label, x_unlabel, y_label, y_test_hat, x_test, y_test\nx_train = np.concatenate([x_label, x_unlabel], axis=0)\nprint(\"x_train shape:\", x_train.shape)\n\ny_label_converted = np.argmax(y_label, axis=1).reshape(-1, 1)\ny_train = np.concatenate([y_label_converted, y_test_hat.reshape(-1, 1)], axis=0)\nprint(\"y_train shape:\", y_train.shape)\n\n# Ghi x_train ra file\nnp.savetxt(\"x_train.csv\", x_train, delimiter=\",\", fmt=\"%.6f\")\nprint(\"x_train.csv đã được ghi thành công!\")\n\n# Ghi y_train ra file\nnp.savetxt(\"y_train.csv\", y_train, delimiter=\",\", fmt=\"%.6f\")\nprint(\"y_train.csv đã được ghi thành công!\")\n\nnp.savetxt(\"test.csv\", test, delimiter=\",\", fmt=\"%.6f\")\nprint(\"test.csv đã được ghi thành công!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:17:36.796041Z","iopub.execute_input":"2024-12-22T15:17:36.796343Z","iopub.status.idle":"2024-12-22T15:17:36.957793Z","shell.execute_reply.started":"2024-12-22T15:17:36.796319Z","shell.execute_reply":"2024-12-22T15:17:36.956939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:17:37.028002Z","iopub.execute_input":"2024-12-22T15:17:37.028256Z","iopub.status.idle":"2024-12-22T15:17:37.031873Z","shell.execute_reply.started":"2024-12-22T15:17:37.028235Z","shell.execute_reply":"2024-12-22T15:17:37.031023Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MODEL","metadata":{}},{"cell_type":"markdown","source":"# Model Training and Evaluation\n\n- **Model Types**: Various models are used, including:\n  - **LightGBM**: A gradient-boosting framework known for its speed and efficiency with large datasets.\n  - **XGBoost**: Another powerful gradient-boosting model used for structured data.\n  - **CatBoost**: Optimized for categorical features without the need for extensive preprocessing.\n  - **Voting Regressor**: An ensemble model that combines the predictions of LightGBM, XGBoost, and CatBoost for better accuracy.\n- **Cross-Validation**: Stratified K-Folds cross-validation is employed to split the data into training and validation sets, ensuring balanced class distribution in each fold.\n- **Quadratic Weighted Kappa (QWK)**: The performance of the models is evaluated using QWK, which measures the agreement between predicted and actual values, taking into account the ordinal nature of the target variable.\n- **Threshold Optimization**: The `minimize` function from `scipy.optimize` is used to fine-tune decision thresholds that map continuous predictions to discrete categories (None, Mild, Moderate, Severe).\n","metadata":{}},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T15:17:39.480715Z","iopub.execute_input":"2024-12-22T15:17:39.480999Z","iopub.status.idle":"2024-12-22T15:17:39.489963Z","shell.execute_reply.started":"2024-12-22T15:17:39.480978Z","shell.execute_reply":"2024-12-22T15:17:39.489021Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# Hyperparameter Tuning\n\n- **LightGBM Parameters**: Hyperparameters such as `learning_rate`, `max_depth`, `num_leaves`, and `feature_fraction` are tuned to improve the performance of the LightGBM model. These parameters control the complexity of the model and its ability to generalize to new data.\n- **XGBoost and CatBoost Parameters**: Similar tuning is applied for XGBoost and CatBoost, adjusting parameters such as `n_estimators`, `max_depth`, `learning_rate`, `subsample`, and `regularization` terms (`reg_alpha`, `reg_lambda`). These help in controlling overfitting and ensuring the model's robustness.","metadata":{}},{"cell_type":"markdown","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) # New","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, thresholds)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),    \n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission = TrainML(ensemble, test)\nSubmission = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission\n})\n\nSubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.idle":"2024-12-22T14:42:44.623142Z","shell.execute_reply.started":"2024-12-22T14:38:20.517208Z","shell.execute_reply":"2024-12-22T14:42:44.621893Z"}},"outputs":[],"execution_count":null}]}