{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":1003630,"sourceType":"datasetVersion","datasetId":442595},{"sourceId":2660070,"sourceType":"datasetVersion","datasetId":686792},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302},{"sourceId":10161923,"sourceType":"datasetVersion","datasetId":6275040},{"sourceId":10162258,"sourceType":"datasetVersion","datasetId":6275292}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:12.74567Z","iopub.execute_input":"2024-12-22T10:16:12.746968Z","iopub.status.idle":"2024-12-22T10:16:16.76534Z","shell.execute_reply.started":"2024-12-22T10:16:12.746926Z","shell.execute_reply":"2024-12-22T10:16:16.763542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --no-index --no-deps /kaggle/input/pytorchtransformer/tab_transformer_pytorch-0.3.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:16.767525Z","iopub.execute_input":"2024-12-22T10:16:16.767995Z","iopub.status.idle":"2024-12-22T10:16:18.008385Z","shell.execute_reply.started":"2024-12-22T10:16:16.767948Z","shell.execute_reply":"2024-12-22T10:16:18.007169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --no-index --no-deps /kaggle/input/pytorch/einops-0.8.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:18.011571Z","iopub.execute_input":"2024-12-22T10:16:18.011904Z","iopub.status.idle":"2024-12-22T10:16:19.089981Z","shell.execute_reply.started":"2024-12-22T10:16:18.011873Z","shell.execute_reply":"2024-12-22T10:16:19.088818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:19.092334Z","iopub.execute_input":"2024-12-22T10:16:19.092819Z","iopub.status.idle":"2024-12-22T10:16:19.103194Z","shell.execute_reply.started":"2024-12-22T10:16:19.092709Z","shell.execute_reply":"2024-12-22T10:16:19.101868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data processing\n","metadata":{}},{"cell_type":"code","source":"## Time-series data statistics extraction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:19.104395Z","iopub.execute_input":"2024-12-22T10:16:19.104737Z","iopub.status.idle":"2024-12-22T10:16:19.117913Z","shell.execute_reply.started":"2024-12-22T10:16:19.104694Z","shell.execute_reply":"2024-12-22T10:16:19.116936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def dictionary_of_statistics(data, time = None):\n    \n    #Aggreate statistics for the dictionary\n    stats_summary = data.agg(['mean', 'median', 'max', 'std']).to_dict()\n    \n    flattened_stats = {}\n    for col, stats in stats_summary.items():\n        for stat_name, value in stats.items():\n            key = f\"{stat_name}_{col}\"\n            if (time is not None):\n                key = f\"{stat_name}_{col}_{time}\"\n            flattened_stats[key] = value\n    \n    return flattened_stats\n\n#Feature engineering\n\ndef compute_time_features(data, day_start_hour=6, day_end_hour=18, expected_diff=5):\n    \n    \"\"\"\n    Compute and add time-related features to the DataFrame.\n\n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'time_of_day' in nanosecond and 'relative_date_PCIAT'.\n    - day_start_hour (int): Hour to start the day period. Default is 8.\n    - day_end_hour (int): Hour to end the day period. Default is 21.\n    - expected_diff (int): Expected time difference between steps in seconds. Default is 5.\n    \n    \"\"\"\n    \n    #From nanosecond to hour in a day\n    data['time_of_day_hours'] = data['time_of_day'] / 1e9 / 3600\n    data['day_time'] = data['relative_date_PCIAT'] + data['time_of_day_hours'] / 24\n    \n    #Categorize the day and night based on time data\n    \n    data['day_period'] = np.where(\n        (data['time_of_day_hours'] >= day_start_hour) &\n        (data['time_of_day_hours'] < day_end_hour),\n        'day', 'night'\n    )\n    \n    #Time difference beween steps\n    #As the description, the time_of_day should represent the start of a 5s window over which the data was sampled\n    #Calculate the time difference between each step\n    data['time_diff'] = (data['day_time'].diff() * 86400).round(0) # seconds in a day\n    data['measurement_after_gap'] = data['time_diff'] > expected_diff\n    \ndef no_motion_periods(worn_data):\n    \"\"\"\n    Find periods of no motion and give analytical insights in the data.\n\n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'time_of_day' and 'relative_date_PCIAT'.\n\n    Returns:\n    - pd.DataFrame: DataFrame with new features: \n    + total duration of no motion periods per day\n    + the number of no motion periods per day.\n    \"\"\"\n    \n    #Calculate no motion periods\n    no_motion = worn_data['enmo'] == 0\n    motion_group = (\n        (no_motion != no_motion.shift()) |\n        (worn_data['measurement_after_gap'])\n    ).cumsum()\n\n    no_motion_periods = worn_data[no_motion].groupby(\n        motion_group\n    )['day_time'].agg(['min', 'max'])\n\n    no_motion_periods['duration_sec'] = (\n        (no_motion_periods['max'] - no_motion_periods['min']) * 86400\n    ).round(0).astype(int)\n    \n    no_motion_periods['duration_sec'] += 5\n    no_motion_periods['day'] = no_motion_periods['min'].astype(int)\n    \n    # Calculate daily statistics on no motion periods\n    daily_stats = no_motion_periods.groupby(no_motion_periods['day']) \\\n        .agg(no_motion_duration=('duration_sec', 'sum'),\n            no_motion_count=('duration_sec', 'size'))\n    \n    #Aggreate statistics for the dictionary\n    return dictionary_of_statistics(daily_stats)\n\ndef circadian_rhythm_analysis(worn_data):\n    \n    \"\"\"\n    Make features capturing the variation in activity across the 24-hour cycle, \n    separately for day and night times (or wakefulness and sleep periods).\n    \n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'time_of_day_hours', 'day_period' and 'relative_date_PCIAT'.\n    \n    Returns:\n    - pd.DataFrame: 2 DataFrame, corresponding to day and night times,  with new features capturing the circadian rhythm of the wearer,\n    + Standard deviation across hourly means per day\n    + Peak hour of activity per day\n    + Entropy of activity distribution per day\n    \"\"\"\n    \n    hourly_activity = worn_data.groupby(\n        [worn_data['relative_date_PCIAT'].astype(int),\n        worn_data['time_of_day_hours'].astype(int),\n        worn_data['day_period']]\n    )['enmo'].agg(['mean', 'max'])\n\n    features = hourly_activity['mean'].groupby(\n        ['relative_date_PCIAT', 'day_period']\n    ).agg(\n        std_across_hours='std',\n        peak_hour=lambda x: x.idxmax()[1],\n        entropy=lambda x: -(x / x.sum() * np.log(x / x.sum() + 1e-9)).sum()\n    )\n    \n    day_features = features.xs('day', level='day_period')\n    night_features = features.xs('night', level='day_period')\n    \n    return dictionary_of_statistics(day_features, time=\"day\") | dictionary_of_statistics(night_features, time='night')\n    # return features\n\ndef physical_activity_analysis(worn_data):\n        \n    \"\"\"\n    Analyze the Moderate to Vigorous Physical Activity (MVPA) based on a threshold of ENMO values, \n    and calculate the duration of the detected MVPA activity bouts\n    \n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'enmo', 'time_diff' and 'day_time'.\n    \n    Returns:\n    - pd.DataFrame: DataFrame with new features capturing the physical activity level of the wearer,\n    including:\n    + Total duration of MVPA per day\n    + Number of MVPA periods per day\n    \"\"\"\n    # In order to classify physical activity as MVPA, we only retained activities that lasted at least 1 minute and met the criteria for the 100 mg (= 0.1g) threshold\n    mvpa_threshold = 0.1\n    merge_gap = 60\n    \n    def merge_mvpa_groups(df, allowed_gap=60, merge_gap=60):\n        last_mvpa_time = df['day_time'].where(df['is_mvpa']).ffill().shift()\n        \n        mvpa_time_diff = (\n            (df['day_time'] - last_mvpa_time) * 86400\n        ).round(0)\n        \n        mvpa_group = (\n            (df['is_mvpa'] != df['is_mvpa'].shift()) |\n            (df['time_diff'] >= allowed_gap)\n        ).cumsum()\n        \n        is_mvpa_start = (\n            (mvpa_group != mvpa_group.shift()) &\n            df['is_mvpa']\n        )\n        \n        group_increment = is_mvpa_start & (\n            (mvpa_time_diff >= merge_gap) | last_mvpa_time.isnull()\n        )\n        \n        merged_group = group_increment.cumsum()\n        merged_group.loc[~df['is_mvpa']] = np.nan\n        \n        return merged_group\n    \n    worn_data['is_mvpa'] = worn_data['enmo'] > mvpa_threshold\n    worn_data['mvpa_merged_group'] = merge_mvpa_groups(worn_data)\n\n    mvpa_periods = worn_data[\n        worn_data['is_mvpa']\n    ].groupby('mvpa_merged_group')['day_time'].agg(['min', 'max'])\n\n    mvpa_periods['duration_sec'] = (\n        mvpa_periods['max'] - mvpa_periods['min']\n    ) * 86400  # days to seconds\n\n    mvpa_periods = mvpa_periods[mvpa_periods['duration_sec'] >= 60]\n    mvpa_periods['duration_min'] = mvpa_periods['duration_sec'] / 60\n    \n    mvpa_periods['day'] = mvpa_periods['min'].astype(int)\n\n    daily_stats = mvpa_periods.groupby(mvpa_periods['day']) \\\n        .agg(mvpa_total_duration=('duration_sec', 'sum'),\n            mvpa_count_periods=('duration_sec', 'size'))\n        \n    return dictionary_of_statistics(daily_stats)\n    \ndef activity_transition_analysis(worn_data):\n        \n    \"\"\"\n    The analysis to look at transitions between low, moderate and vigorous activity. To smooth out sudden, \n    short bursts of different activities, we filter out segments with a duration below a 1 minute threshold.\n    \n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'enmo', 'time_diff' and 'day_time'.\n    \n    Returns:\n    - pd.DataFrame: DataFrame with new features capturing the activity transitions of the wearer,\n    + Total duration of different of activity per day\n    + Number of different activity periods per day\n    \"\"\"\n    mvpa_threshold = 0.1\n    vig_threshold = 0.5\n    worn_data['activity_type'] = pd.cut(\n        worn_data['enmo'],\n        bins=[-np.inf, mvpa_threshold, vig_threshold, np.inf],\n        labels=['low', 'moderate', 'vigorous']\n    )\n    activity_group = (\n        (worn_data['activity_type'] != worn_data['activity_type'].shift()) |\n        (worn_data['measurement_after_gap'])\n    ).cumsum()\n    \n    activity_periods = worn_data.groupby(activity_group).agg(\n        min=('day_time', 'min'),\n        max=('day_time', 'max'),\n        activity_type=('activity_type', 'first')\n    )\n    activity_periods['duration_sec'] = (\n        activity_periods['max'] - activity_periods['min']\n    ) * 86400 + 5 \n\n    activity_periods = activity_periods[activity_periods['duration_sec'] >= 60]\n    activity_periods['duration_min'] = activity_periods['duration_sec'] / 60\n    \n    activity_periods['day'] = activity_periods['min'].astype(int)\n    activity_periods['transition_num'] = (\n        activity_periods.groupby('day')['activity_type']\n        .apply(lambda x: (x != x.shift()).cumsum())\n        .reset_index(level=0, drop=True)\n    )\n    \n    low_activity = activity_periods[activity_periods['activity_type'] == 'low'].groupby('day').agg(\n        low_act_total_duration=('duration_sec', 'sum'),\n        low_act_count_periods=('duration_sec', 'size')\n    )\n    \n    moderate_activity = activity_periods[activity_periods['activity_type'] == 'moderate'].groupby('day').agg(\n            moderate_act_total_duration=('duration_sec', 'sum'),\n            moderate_act_count_periods=('duration_sec', 'size')\n    )\n\n        \n    daily_transitions = activity_periods.groupby('day').agg(\n        count_transitions=('transition_num', 'max')\n    )\n    \n    return dictionary_of_statistics(low_activity) | dictionary_of_statistics(moderate_activity) | dictionary_of_statistics(daily_transitions)\n\ndef activity_light_exposure(worn_data):\n        \n    \"\"\"\n    Analyze the correlation between light exposure and physical activity level.\n    \n    Parameters:\n    - data (pd.DataFrame): The input DataFrame containing 'light' and 'enmo'.\n    \n    Returns:\n    - float: The correlation between light exposure and physical activity level.\n    \"\"\"\n    correlation_light_enmo = worn_data[['light', 'enmo']].corr().iloc[0, 1]\n    return {'correlation_light_enmo': correlation_light_enmo}\n\ndef process_file(file_path, participant_id):\n    data = pd.read_parquet(file_path)\n\n    # Compute time features\n    compute_time_features(data)\n        \n    # Calculate the percentage of non-worn time\n    non_wear_percentage = (data['non-wear_flag'].sum() / len(data)) * 100\n    \n    # Filter out the worn data\n    worn_data = data[data['non-wear_flag'] == 0]\n    \n    # recalculate time difference between rows and measurement_after_gap flag in the worn data\n    expected_diff = 5\n    worn_data['time_diff'] = (worn_data['day_time'].diff() * 86400).round(0)\n    worn_data['measurement_after_gap'] = worn_data['time_diff'] > expected_diff\n\n    # Compute no motion periods\n    no_motion_stats = no_motion_periods(worn_data)\n    # Circadian rhythm analysis\n    circadian_rhythm_stats = circadian_rhythm_analysis(worn_data)\n    # Physical activity analysis\n    physical_activity_stats = physical_activity_analysis(worn_data)\n    # Activity transition analysis\n    activity_transition_stats = activity_transition_analysis(worn_data)\n    # Compute activity light correlation\n    activity_light_stats = activity_light_exposure(worn_data)\n\n    return {\n        'id': participant_id,\n    } | no_motion_stats | circadian_rhythm_stats | physical_activity_stats | activity_transition_stats | activity_light_stats\n\ndef load_time_series(dir_name):\n    \n    participant_ids = os.listdir(dir_name)\n    \n    with ThreadPoolExecutor() as executor:\n        #tqdm: Wraps the executor.map iterable with tqdm -> Show a progress bar indicating the processing status\n        results = list(tqdm(executor.map(lambda x: process_file(os.path.join(dir_name, x, 'part-0.parquet'), x), participant_ids), total=len(participant_ids)))\n\n    df = pd.DataFrame(results)\n    # Replace inf and -inf with NaN\n    df.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n    # Replace NaN with 0\n    df.fillna(0, inplace=True)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:19.119322Z","iopub.execute_input":"2024-12-22T10:16:19.119722Z","iopub.status.idle":"2024-12-22T10:16:19.151646Z","shell.execute_reply.started":"2024-12-22T10:16:19.11969Z","shell.execute_reply":"2024-12-22T10:16:19.150613Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Build a AutoEncoder to compress size of time-series dataset, helps to remove redundant features and avoid overfitting**","metadata":{}},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:19.152803Z","iopub.execute_input":"2024-12-22T10:16:19.153119Z","iopub.status.idle":"2024-12-22T10:16:19.171335Z","shell.execute_reply.started":"2024-12-22T10:16:19.153077Z","shell.execute_reply":"2024-12-22T10:16:19.170328Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"markdown","source":"**Loading tabular data**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:19.175451Z","iopub.execute_input":"2024-12-22T10:16:19.17586Z","iopub.status.idle":"2024-12-22T10:16:19.248475Z","shell.execute_reply.started":"2024-12-22T10:16:19.175828Z","shell.execute_reply":"2024-12-22T10:16:19.247417Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Few SII scores are still derived from the sum of NAN values in PICAT questions, leading to potentially invalid SII values. The below code tries to estimate the severity of internet usage SII based on current SII and the maximum possible SII**","metadata":{}},{"cell_type":"code","source":"#Generate a list of column names in the format PCIAT-PCIAT_XX\nPCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\n\n#Recalculates the SII value based on the the current PCIAT values and the possible maximum PCIAT values\ndef recalculate_sii(row):\n    value = 0\n    if (not pd.isna(row['PCIAT-PCIAT_Total'])):\n        value = row['PCIAT-PCIAT_Total']\n        \n    max_possible = value + row[PCIAT_cols].isna().sum() * 5\n    \n    if value <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= value <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= value <= 79 and max_possible <= 79:\n        return 2\n    elif value >= 80 and max_possible >= 80:\n        return 3\n    \n    return np.nan\n\ntrain['recalc_sii'] = train.apply(recalculate_sii, axis=1)\ntrain['sii'] = train['recalc_sii']\ntrain.drop(columns='recalc_sii', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:19.250156Z","iopub.execute_input":"2024-12-22T10:16:19.250461Z","iopub.status.idle":"2024-12-22T10:16:21.259759Z","shell.execute_reply.started":"2024-12-22T10:16:19.250436Z","shell.execute_reply":"2024-12-22T10:16:21.25888Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Loading time-series data**","metadata":{}},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:16:21.260752Z","iopub.execute_input":"2024-12-22T10:16:21.261091Z","iopub.status.idle":"2024-12-22T10:20:49.001595Z","shell.execute_reply.started":"2024-12-22T10:16:21.261052Z","shell.execute_reply":"2024-12-22T10:20:49.000471Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Perform auto encoding on the time_series data**","metadata":{}},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:20:49.002848Z","iopub.execute_input":"2024-12-22T10:20:49.003162Z","iopub.status.idle":"2024-12-22T10:20:59.585386Z","shell.execute_reply.started":"2024-12-22T10:20:49.003135Z","shell.execute_reply":"2024-12-22T10:20:59.584078Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Take a look at the time-series data before and after performing auto encoding. There is a minimal difference in data dimensionality, which means that the feature engineering approach applied to this dataset is effective**","metadata":{}},{"cell_type":"code","source":"print(df_train.shape, train_ts_encoded.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:20:59.586521Z","iopub.execute_input":"2024-12-22T10:20:59.586924Z","iopub.status.idle":"2024-12-22T10:20:59.592624Z","shell.execute_reply.started":"2024-12-22T10:20:59.586886Z","shell.execute_reply":"2024-12-22T10:20:59.591875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain_ts_encoded['id'] = train_ts_encoded['id'].str.replace('id=', '')\ntest_ts_encoded['id'] = test_ts_encoded['id'].str.replace('id=', '')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:20:59.593233Z","iopub.execute_input":"2024-12-22T10:20:59.593505Z","iopub.status.idle":"2024-12-22T10:20:59.611553Z","shell.execute_reply.started":"2024-12-22T10:20:59.593481Z","shell.execute_reply":"2024-12-22T10:20:59.61036Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data merging","metadata":{}},{"cell_type":"code","source":"train = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:20:59.612728Z","iopub.execute_input":"2024-12-22T10:20:59.613138Z","iopub.status.idle":"2024-12-22T10:20:59.63889Z","shell.execute_reply.started":"2024-12-22T10:20:59.613107Z","shell.execute_reply":"2024-12-22T10:20:59.637872Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Take a look at the test set after grafting encoded time-series data. As shown below, the participants who didn't wear device have features related to time-series values are NaN. To handle, we will replace all NaN values with 0**","metadata":{}},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:20:59.640193Z","iopub.execute_input":"2024-12-22T10:20:59.6406Z","iopub.status.idle":"2024-12-22T10:20:59.73014Z","shell.execute_reply.started":"2024-12-22T10:20:59.640568Z","shell.execute_reply":"2024-12-22T10:20:59.728937Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handle missing in predictors","metadata":{}},{"cell_type":"code","source":"train_cols = set(train.columns)\ntest_cols = set(test.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\ncolumns_not_in_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:20:59.731283Z","iopub.execute_input":"2024-12-22T10:20:59.73156Z","iopub.status.idle":"2024-12-22T10:20:59.738751Z","shell.execute_reply.started":"2024-12-22T10:20:59.731535Z","shell.execute_reply":"2024-12-22T10:20:59.737628Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Initialize a KNNImputer for handling missing valuee. Statistically, KNNImputer outweights SimpleImputer in this context. However, it encounters the computationally expensive cost, and the requirement of hyperparameter K (K value should be considered to change later)**","metadata":{}},{"cell_type":"code","source":"# Get numeric columns excluding 'sii'\nnumeric_cols = train.select_dtypes(include=['float64', 'int64', 'int32', 'float32']).columns.tolist()\nnumeric_cols.remove('sii')\n\n# Handle numerical with KNNImputer\nimputer = KNNImputer(n_neighbors=5)\ntrain[numeric_cols] = imputer.fit_transform(train[numeric_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:20:59.7399Z","iopub.execute_input":"2024-12-22T10:20:59.74022Z","iopub.status.idle":"2024-12-22T10:21:14.328931Z","shell.execute_reply.started":"2024-12-22T10:20:59.740195Z","shell.execute_reply":"2024-12-22T10:21:14.327896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Get categorical columns\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\n#File the nan values in categorical collumns with 'Missing'\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\n#Convert from categorical to numeric \ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.329971Z","iopub.execute_input":"2024-12-22T10:21:14.330325Z","iopub.status.idle":"2024-12-22T10:21:14.392878Z","shell.execute_reply.started":"2024-12-22T10:21:14.33029Z","shell.execute_reply":"2024-12-22T10:21:14.392029Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FEATURE ENGINEERING","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n\n    # Replace any remaining inf values with NaN\n    df = df.replace([np.inf, -np.inf], np.nan)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.393636Z","iopub.execute_input":"2024-12-22T10:21:14.393938Z","iopub.status.idle":"2024-12-22T10:21:14.400998Z","shell.execute_reply.started":"2024-12-22T10:21:14.393914Z","shell.execute_reply":"2024-12-22T10:21:14.399891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = feature_engineering(train)\n#Remove rows which has got fewer than 10 non-null values\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.402327Z","iopub.execute_input":"2024-12-22T10:21:14.402713Z","iopub.status.idle":"2024-12-22T10:21:14.475325Z","shell.execute_reply.started":"2024-12-22T10:21:14.402661Z","shell.execute_reply":"2024-12-22T10:21:14.474287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii',\n                'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW','BMI_PHR']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\n\nfeaturesCols.remove('sii')\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.476442Z","iopub.execute_input":"2024-12-22T10:21:14.476726Z","iopub.status.idle":"2024-12-22T10:21:14.486327Z","shell.execute_reply.started":"2024-12-22T10:21:14.4767Z","shell.execute_reply":"2024-12-22T10:21:14.485439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Check and replace any INF values with NaN\nif np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.487344Z","iopub.execute_input":"2024-12-22T10:21:14.487721Z","iopub.status.idle":"2024-12-22T10:21:14.509729Z","shell.execute_reply.started":"2024-12-22T10:21:14.487682Z","shell.execute_reply":"2024-12-22T10:21:14.508447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = train\n# Get columns with NaN values\nnan_columns = df.columns[df.isna().any()].tolist()\n\n# Calculate number of NaN values per column\nnan_counts = df[nan_columns].isna().sum()\n\n# Sort by number of NaN values (descending)\nnan_counts_sorted = nan_counts.sort_values(ascending=False)\n\n# Display results\nprint(\"\\nColumns with NaN values:\")\nprint(\"-\" * 50)\nfor col, count in nan_counts_sorted.items():\n    total = len(df)\n    percentage = (count/total * 100)\n    print(f\"{col:<30} {count:>7} NaN values ({percentage:>6.2f}%)\")\n\nprint(f\"\\nTotal columns with NaN values: {len(nan_columns)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.515286Z","iopub.execute_input":"2024-12-22T10:21:14.515609Z","iopub.status.idle":"2024-12-22T10:21:14.527637Z","shell.execute_reply.started":"2024-12-22T10:21:14.515583Z","shell.execute_reply":"2024-12-22T10:21:14.52652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.dropna(subset=['BMR_Weight', 'DEE_Weight', 'Hydration_Status'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.529646Z","iopub.execute_input":"2024-12-22T10:21:14.530008Z","iopub.status.idle":"2024-12-22T10:21:14.544541Z","shell.execute_reply.started":"2024-12-22T10:21:14.529977Z","shell.execute_reply":"2024-12-22T10:21:14.543456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train.to_csv(\"train1.csv\", index=False)\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.545676Z","iopub.execute_input":"2024-12-22T10:21:14.546037Z","iopub.status.idle":"2024-12-22T10:21:14.561424Z","shell.execute_reply.started":"2024-12-22T10:21:14.546009Z","shell.execute_reply":"2024-12-22T10:21:14.560434Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# VIME Definition\n## Mask & Pretext generator\n**The below cell defines severals utilities for building VIME, including mask_generator & pretext_generator**","metadata":{}},{"cell_type":"code","source":"def mask_generator (p_m, x):\n  \"\"\"Generate mask vector.\n  \n  Args:\n    - p_m: corruption probability\n    - x: feature matrix\n    \n  Returns:\n    - mask: binary mask matrix \n  \"\"\"\n  mask = np.random.binomial(1, p_m, x.shape)\n  return mask\n\ndef pretext_generator (m, x):  \n  \"\"\"Generate corrupted samples.\n  \n  Args:\n    m: mask matrix\n    x: feature matrix\n    \n  Returns:\n    m_new: final mask matrix after corruption\n    x_tilde: corrupted feature matrix\n  \"\"\"\n  \n  # Parameters\n  no, dim = x.shape  \n  # Randomly (and column-wise) shuffle data\n  x_bar = np.zeros([no, dim])\n  for i in range(dim):\n    idx = np.random.permutation(no)\n    x_bar[:, i] = x[idx, i]\n    \n  # Corrupt samples\n  x_tilde = x * (1-m) + x_bar * m  \n  # Define new mask matrix\n  m_new = 1 * (x != x_tilde)\n\n  return m_new, x_tilde\n\ndef convert_matrix_to_vector(matrix):\n  \"\"\"Convert two dimensional matrix into one dimensional vector\n  \n  Args:\n    - matrix: two dimensional matrix\n    \n  Returns:\n    - vector: one dimensional vector\n  \"\"\"\n  # Parameters\n  no, dim = matrix.shape\n  # Define output  \n  vector = np.zeros([no,])\n  \n  # Convert matrix to vector\n  for i in range(dim):\n    idx = np.where(matrix[:, i] == 1)\n    vector[idx] = i\n    \n  return vector\n\ndef convert_vector_to_matrix(vector):\n  \"\"\"Convert one dimensional vector into two dimensional matrix\n  \n  Args:\n    - vector: one dimensional vector\n    \n  Returns:\n    - matrix: two dimensional matrix\n  \"\"\"\n  # Parameters\n  no = len(vector)\n  dim = len(np.unique(vector))\n  # Define output\n  matrix = np.zeros([no,dim])\n  \n  # Convert vector to matrix\n  for i in range(dim):\n    idx = np.where(vector == i)\n    matrix[idx, i] = 1\n    \n  return matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.562498Z","iopub.execute_input":"2024-12-22T10:21:14.562831Z","iopub.status.idle":"2024-12-22T10:21:14.579484Z","shell.execute_reply.started":"2024-12-22T10:21:14.56277Z","shell.execute_reply":"2024-12-22T10:21:14.578317Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## VIME Self-supervised framework","metadata":{}},{"cell_type":"code","source":"# Necessary packages\nfrom tensorflow import keras\nfrom tensorflow.keras import Model, Input, layers\n\ndef vime_self (x_unlab, p_m, alpha, parameters):\n  \"\"\"Self-supervised learning part in VIME.\n  \n  Args:\n    x_unlab: unlabeled feature\n    p_m: corruption probability\n    alpha: hyper-parameter to control the weights of feature and mask losses\n    parameters: epochs, batch_size\n    \n  Returns:\n    encoder: Representation learning block\n  \"\"\"\n    \n  # Parameters\n  _, dim = x_unlab.shape\n  epochs = parameters['epochs']\n  batch_size = parameters['batch_size']\n  \n  # Build model using Functional API\n  inputs = Input(shape=(dim,))\n  # Encoder\n  h = layers.Dense(dim, activation='relu')(inputs)\n  # Mask estimator\n  mask_output = layers.Dense(dim, activation='sigmoid', name='mask')(h)\n  # Feature estimator\n  feature_output = layers.Dense(dim, activation='sigmoid', name='feature')(h)\n  \n  #Create model\n  model = Model(inputs = inputs, outputs = [mask_output, feature_output])\n  \n  model.compile(optimizer='rmsprop',\n                loss={'mask': 'binary_crossentropy', \n                      'feature': 'mean_squared_error'},\n                loss_weights={'mask':1.0\n                              , 'feature':float(alpha)})\n  \n  # Generate corrupted samples\n  m_unlab = mask_generator(p_m, x_unlab)\n  m_label, x_tilde = pretext_generator(m_unlab, x_unlab)\n  \n  # Fit model on unlabeled data\n  model.fit(x_tilde, {'mask': m_label, 'feature': x_unlab}, \n            epochs = epochs, batch_size= batch_size)\n      \n  # Extract encoder\n  encoder = Model(\n      inputs=model.input,\n      outputs=model.layers[1].output,\n      name='encoder'\n  )\n  \n  return encoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.580612Z","iopub.execute_input":"2024-12-22T10:21:14.581001Z","iopub.status.idle":"2024-12-22T10:21:14.650734Z","shell.execute_reply.started":"2024-12-22T10:21:14.58096Z","shell.execute_reply":"2024-12-22T10:21:14.649582Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## VIME Semi-supervised framework","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nimport numpy as np\nfrom tensorflow import keras\n \n\"\"\"\nExpected flow:\n- Load & process data\n- Create network architecture\n- Train the model using:\n+ Supervised loss from labeled data\n+ Unsupervised loss from unlabeled data\n\"\"\"\n\nclass Predictor(keras.Model):\n    def __init__(self, hidden_dim, label_dim):\n        super(Predictor, self).__init__()\n        #Define layers\n        self.dense1 = keras.layers.Dense(hidden_dim, activation='relu')\n        self.dense2 = keras.layers.Dense(hidden_dim, activation='relu')\n        self.output_layer = keras.layers.Dense(label_dim)\n        \n    def call(self, x_input):\n        #Forward pass\n        inter_layer = self.dense1(x_input)\n        inter_layer = self.dense2(inter_layer)\n        y_hat_logit = self.output_layer(inter_layer)\n        y_hat = tf.nn.softmax(y_hat_logit)\n        return y_hat_logit, y_hat\n \ndef vime_semi(x_train, y_train, x_unlab, x_test, parameters, \n              p_m, K, beta, file_name):\n    \"\"\"Semi-supervied learning part in VIME.\n    \n    Args:\n        - x_train, y_train: training dataset\n        - x_unlab: unlabeled dataset\n        - x_test: testing features\n        - parameters: network parameters (hidden_dim, batch_size, iterations)\n        - p_m: corruption probability\n        - K: number of augmented samples\n        - beta: hyperparameter to control supervised and unsupervised loss\n        - file_name: saved filed name for the encoder function\n        \n    Returns:\n        - y_test_hat: prediction on x_test\n    \"\"\"\n    class WeightedKappaLoss(tf.keras.losses.Loss):\n        def __init__(self, num_classes, name=\"weighted_kappa_loss\"):\n            super().__init__(name=name)\n            self.num_classes = num_classes\n            # Create weight matrix\n            weights = tf.cast(tf.range(num_classes), tf.float32)\n            weights = tf.expand_dims(weights, 0) - tf.expand_dims(weights, 1)\n            self.weights = tf.square(weights)\n            \n        @tf.function\n        def call(self, y_true, y_pred):\n            # Convert types\n            y_true = tf.cast(y_true, tf.float32)\n            y_pred = tf.cast(y_pred, tf.float32)\n            \n            # Apply softmax\n            y_pred = tf.nn.softmax(y_pred)\n            \n            # Calculate confusion matrix\n            batch_size = tf.cast(tf.shape(y_true)[0], tf.float32)\n            confusion = tf.matmul(y_true, y_pred, transpose_a=True)\n            confusion = confusion / batch_size\n            \n            # Calculate agreements\n            observed = tf.reduce_sum(confusion * (1.0 - self.weights))\n            expected_rows = tf.reduce_sum(confusion, axis=1)\n            expected_cols = tf.reduce_sum(confusion, axis=0)\n            expected = tf.reduce_sum(\n                tf.matmul(tf.expand_dims(expected_rows, 1),\n                         tf.expand_dims(expected_cols, 0)) * (1.0 - self.weights)\n            )\n            \n            # Calculate kappa\n            kappa = (observed - expected) / (1.0 - expected + 1e-8)\n            return 1.0 - kappa\n    \n    # Network parameters\n    hidden_dim = parameters['hidden_dim']\n    batch_size = parameters['batch_size']\n    iterations = parameters['iterations']\n      \n    # Basic parameters\n    data_dim = x_train.shape[1]\n    label_dim = y_train.shape[1]\n\n    # Divide training and validation sets (9:1)\n    idx = np.random.permutation(len(x_train))\n    train_idx = idx[:int(len(idx)*0.9)]\n    valid_idx = idx[int(len(idx)*0.9):]\n    \n    x_valid, y_valid = x_train[valid_idx], y_train[valid_idx]\n    x_train, y_train = x_train[train_idx], y_train[train_idx]\n    \n    # Load encoder from self-supervised model\n    encoder = keras.models.load_model(file_name)\n    \n    # Create predictor model\n    predictor = Predictor(hidden_dim, label_dim)\n    optimizer = keras.optimizers.Adam()\n    \n    # Encode validation and testing features\n    x_valid = encoder.predict(x_valid)\n    x_test = encoder.predict(x_test)\n    \n    # Setup checkpointing\n    checkpoint = tf.train.Checkpoint(model=predictor)\n    manager = tf.train.CheckpointManager(\n        checkpoint, './save_model', max_to_keep=1)\n    \n    # Early stopping variables\n    best_valid_loss = float('inf')\n    patience = -1\n    \n    # Initialize loss\n    loss_fn = WeightedKappaLoss(num_classes=label_dim)\n    \n    @tf.function\n    def train_step(x_batch, y_batch, xu_batch):\n        with tf.GradientTape() as tape:\n            #Forward pass\n            y_logits, _ = predictor(x_batch)\n            yv_logits, _ = predictor(xu_batch)\n            \n            # Calculate supervised loss\n            supervised_loss = loss_fn(y_batch, y_logits)\n\n            # Unsupervised loss using variance\n            unsupervised_loss = tf.reduce_mean(\n                tf.math.reduce_variance(yv_logits, axis=0))\n            \n            total_loss = supervised_loss + beta * unsupervised_loss\n        \n        # Compute gradients and update weights\n        gradients = tape.gradient(total_loss, predictor.trainable_variables)\n        optimizer.apply_gradients(zip(gradients, predictor.trainable_variables))\n        \n        return total_loss\n        \n    for it in range(iterations):\n        # Select a batch of labeled data\n        batch_idx = np.random.permutation(len(x_train))[:batch_size]\n        x_batch = x_train[batch_idx]\n        y_batch = y_train[batch_idx]\n        \n        # Encode labeled data\n        x_batch = encoder.predict(x_batch)\n        \n        # Select and augment unlabeled data\n        batch_u_idx = np.random.permutation(len(x_unlab))[:batch_size]\n        xu_batch_ori = x_unlab[batch_u_idx]\n        \n        xu_batch = []\n        for _ in range(K):\n            # Mask vector generation\n            m_batch = mask_generator(p_m, xu_batch_ori)\n            # Pretext generator\n            _, xu_batch_temp = pretext_generator(m_batch, xu_batch_ori)\n            \n            # Encode corrupted samples\n            xu_batch_temp = encoder.predict(xu_batch_temp)\n            xu_batch.append(xu_batch_temp)\n        \n        #Convert list to matrix\n        xu_batch = tf.convert_to_tensor(np.array(xu_batch))\n        \n        # Training step\n        loss = train_step(x_batch, y_batch, xu_batch)\n        \n        # Validation step\n        val_logits, _ = predictor(x_valid)\n        \n        val_loss = loss_fn(y_valid, val_logits)\n        \n        if it % 100 == 0:\n            print(f'Iteration: {it}/{iterations}, Loss: {loss:.4f}, Val Loss: {val_loss:.4f}')\n            \n        # Early stopping\n        if val_loss < best_valid_loss:\n            best_valid_loss = val_loss\n            manager.save()\n            patience = 0\n        else:\n            patience += 1\n            if patience >= 100:\n                break\n            \n        \n    # Restore best model\n    checkpoint.restore(manager.latest_checkpoint)\n    \n    # Generate predictions\n    _, y_test_hat = predictor(x_test)\n    \n    return y_test_hat.numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.652028Z","iopub.execute_input":"2024-12-22T10:21:14.652402Z","iopub.status.idle":"2024-12-22T10:21:14.673427Z","shell.execute_reply.started":"2024-12-22T10:21:14.652364Z","shell.execute_reply":"2024-12-22T10:21:14.672216Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Labeling","metadata":{}},{"cell_type":"markdown","source":"## Split training to label & unlabel","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\ndef load_data(train):\n    df = train.copy()\n    \n    cols = df.columns\n    categorial_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\n    normalized_cols = [col for col in cols if col not in categorial_cols]\n    normalized_cols.remove('sii')\n    \n    scaler = MinMaxScaler()\n    df[normalized_cols] = scaler.fit_transform(df[normalized_cols])\n    \n    # Split into labeled and unlabeled sets\n    labeled_df = df[df['sii'].notna()]\n    unlabeled_df = df[df['sii'].isna()]\n    \n    x_label = labeled_df.drop(columns=['sii'])\n    y_label = labeled_df['sii']\n    y_label = np.asarray(pd.get_dummies(y_label))\n    x_unlabel = unlabeled_df.drop(columns=['sii'])\n    \n    return x_label, y_label, x_unlabel, normalized_cols, scaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.674597Z","iopub.execute_input":"2024-12-22T10:21:14.674882Z","iopub.status.idle":"2024-12-22T10:21:14.698024Z","shell.execute_reply.started":"2024-12-22T10:21:14.674856Z","shell.execute_reply":"2024-12-22T10:21:14.697042Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Load labeled & unlabeled data, and convert them to numpy.narray type**","metadata":{}},{"cell_type":"code","source":"x_label, y_label, x_unlabel, normalized_cols, scaler = load_data(train)\n\nx_label = x_label.to_numpy().astype(np.float32)\nx_unlabel = x_unlabel.to_numpy().astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.699119Z","iopub.execute_input":"2024-12-22T10:21:14.699517Z","iopub.status.idle":"2024-12-22T10:21:14.757594Z","shell.execute_reply.started":"2024-12-22T10:21:14.69948Z","shell.execute_reply":"2024-12-22T10:21:14.756427Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Apply min-max scaler on test set**","metadata":{}},{"cell_type":"code","source":"test[normalized_cols] = scaler.transform(test[normalized_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.758559Z","iopub.execute_input":"2024-12-22T10:21:14.7589Z","iopub.status.idle":"2024-12-22T10:21:14.790601Z","shell.execute_reply.started":"2024-12-22T10:21:14.758871Z","shell.execute_reply":"2024-12-22T10:21:14.789518Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## VIME Usage","metadata":{}},{"cell_type":"markdown","source":"**Define hyperparameters for VIME**","metadata":{}},{"cell_type":"code","source":"#Define hyperparameters\np_m = 0.3 #Corruption probability for self-supervised learning\nalpha = 2.0 #Control the weights of feature and mask losses\nK = 3 #number of augmented samples\nbeta = 1.0 #Control the weights of the supervised and unsupervised losses","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.791807Z","iopub.execute_input":"2024-12-22T10:21:14.792192Z","iopub.status.idle":"2024-12-22T10:21:14.799162Z","shell.execute_reply.started":"2024-12-22T10:21:14.792162Z","shell.execute_reply":"2024-12-22T10:21:14.798111Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train VIME-Self**","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_rows', 20)  # Show only 20 rows\npd.set_option('display.min_rows', 10)  # Minimum rows to show\n# For Jupyter notebook display\nfrom IPython.display import display_html\n\nvime_self_parameters = dict()\nvime_self_parameters['batch_size'] = 128\nvime_self_parameters['epochs'] = 50\nvime_self_encoder = vime_self(x_unlabel, p_m, alpha, vime_self_parameters)\n\n#Save encoder\nif not os.path.exists('save_model'):\n  os.makedirs('save_model')\n\nfile_name = './save_model/encoder_model.h5'\n  \nvime_self_encoder.save(file_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:14.800341Z","iopub.execute_input":"2024-12-22T10:21:14.800899Z","iopub.status.idle":"2024-12-22T10:21:18.984845Z","shell.execute_reply.started":"2024-12-22T10:21:14.800859Z","shell.execute_reply":"2024-12-22T10:21:18.983931Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train VIME_Semi using encoder from VIME_Self**","metadata":{}},{"cell_type":"code","source":"# Train VIME-Semi\nvime_semi_parameters = dict()\nvime_semi_parameters['hidden_dim'] = 50\nvime_semi_parameters['batch_size'] = 128\nvime_semi_parameters['iterations'] = 1000\n\nfile_name = './save_model/encoder_model.h5'\n\ny_test_hat = vime_semi(x_label, y_label, x_unlabel, x_unlabel, \n                       vime_semi_parameters, p_m, K, beta, file_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T10:21:18.98593Z","iopub.execute_input":"2024-12-22T10:21:18.986295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_hat = np.argmax(y_test_hat, axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count unique values in y_test_hat\nunique_values, value_counts = np.unique(y_test_hat, return_counts=True)\n\n# Display results\nfor value, count in zip(unique_values, value_counts):\n    print(f\"Value {value}: {count} occurrences\")\n\n# Optional: verify total\nprint(f\"\\nTotal elements: {len(y_test_hat)}\")\nprint(f\"Sum of counts: {np.sum(value_counts)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data consolidation","metadata":{}},{"cell_type":"code","source":"# x_train = np.concatenate([x_label, x_unlabel], axis=0)\n# print(x_train.shape)\n# x_train","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# y_label_converted = np.argmax(y_label, axis=1).reshape(-1, 1)\n# y_train = np.concatenate([y_label_converted, y_test_hat.reshape(-1, 1)],axis=0)\n# print(y_train.shape)\n# y_train","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\n# Giả sử các mảng sau đây đã được định nghĩa\n# x_label, x_unlabel, y_label, y_test_hat, x_test, y_test\nx_train = np.concatenate([x_label, x_unlabel], axis=0)\nprint(\"x_train shape:\", x_train.shape)\n\ny_label_converted = np.argmax(y_label, axis=1).reshape(-1, 1)\ny_train = np.concatenate([y_label_converted, y_test_hat.reshape(-1, 1)], axis=0)\nprint(\"y_train shape:\", y_train.shape)\n\n# Ghi x_train ra file\nnp.savetxt(\"x_train.csv\", x_train, delimiter=\",\", fmt=\"%.6f\")\nprint(\"x_train.csv đã được ghi thành công!\")\n\n# Ghi y_train ra file\nnp.savetxt(\"y_train.csv\", y_train, delimiter=\",\", fmt=\"%.6f\")\nprint(\"y_train.csv đã được ghi thành công!\")\n\nnp.savetxt(\"test.csv\", test, delimiter=\",\", fmt=\"%.6f\")\nprint(\"test.csv đã được ghi thành công!\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MODEL","metadata":{}},{"cell_type":"markdown","source":"# Model Training and Evaluation\n\n- **Model Types**: Various models are used, including:\n  - **LightGBM**: A gradient-boosting framework known for its speed and efficiency with large datasets.\n  - **XGBoost**: Another powerful gradient-boosting model used for structured data.\n  - **CatBoost**: Optimized for categorical features without the need for extensive preprocessing.\n  - **Voting Regressor**: An ensemble model that combines the predictions of LightGBM, XGBoost, and CatBoost for better accuracy.\n- **Cross-Validation**: Stratified K-Folds cross-validation is employed to split the data into training and validation sets, ensuring balanced class distribution in each fold.\n- **Quadratic Weighted Kappa (QWK)**: The performance of the models is evaluated using QWK, which measures the agreement between predicted and actual values, taking into account the ordinal nature of the target variable.\n- **Threshold Optimization**: The `minimize` function from `scipy.optimize` is used to fine-tune decision thresholds that map continuous predictions to discrete categories (None, Mild, Moderate, Severe).\n","metadata":{}},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# Hyperparameter Tuning\n\n- **LightGBM Parameters**: Hyperparameters such as `learning_rate`, `max_depth`, `num_leaves`, and `feature_fraction` are tuned to improve the performance of the LightGBM model. These parameters control the complexity of the model and its ability to generalize to new data.\n- **XGBoost and CatBoost Parameters**: Similar tuning is applied for XGBoost and CatBoost, adjusting parameters such as `n_estimators`, `max_depth`, `learning_rate`, `subsample`, and `regularization` terms (`reg_alpha`, `reg_lambda`). These help in controlling overfitting and ensuring the model's robustness.","metadata":{}},{"cell_type":"code","source":"# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'cpu'\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU'\n\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# New: TabNet\n\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        # Create internal validation set\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, \n            y, \n            test_size=0.2,\n            random_state=42\n        )\n        \n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=200,\n            patience=20,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        # Add deepcopy support for scikit-learn\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# TabNet hyperparameters\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Check if current metric is better than best\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)  # Save the entire model","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensemble Learning and Submission Preparation\n\n- **Ensemble Learning**: The model uses a **Voting Regressor**, which combines the predictions from LightGBM, XGBoost, and CatBoost. This approach is beneficial as it leverages the strengths of multiple models, reducing overfitting and improving overall model performance.\n- **Out-of-Fold (OOF) Predictions**: During cross-validation, out-of-fold predictions are generated for the training set, which helps in model evaluation without data leakage.\n- **Kappa Optimizer**: The Kappa Optimizer ensures that the predicted values are as close to the actual values as possible by adjusting the thresholds used to convert raw model outputs into class labels.\n- **Test Set Predictions**: After the model is trained and thresholds are optimized, the test dataset is processed, and predictions are generated using the ensemble model. These predictions are converted into the appropriate format for submission.\n- **Submission File Creation**: The predictions are saved in a CSV file following the required format for submission (e.g., for a Kaggle competition), which includes columns like `id` and `sii` (Severity Impairment Index).","metadata":{}},{"cell_type":"markdown","source":"# Final Results and Performance Metrics\n\n- **Train and Validation Scores**: After training across multiple folds, the mean Quadratic Weighted Kappa (QWK) score is calculated for both the training and validation datasets, providing an indicator of model performance. \n- **Optimized QWK Score**: The final optimized QWK score after threshold tuning is displayed, showcasing the model's ability to predict the severity levels effectively.\n- **Test Predictions**: The test set predictions are evaluated, and a breakdown of the predicted severity levels (None, Mild, Moderate, Severe) is shown, along with their respective counts.","metadata":{}},{"cell_type":"markdown","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) # New","metadata":{}},{"cell_type":"markdown","source":"# **》》》Model1.Train**\n","metadata":{}},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n],weights=[4.0,4.0,5.0,4.0])\n\nSubmission1 = TrainML(voting_model, test)\n\nSubmission1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n# **》》》Model2**\n---","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.49, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    fold_weights = [1.25, 1.0, 1.0, 1.0, 1.0]\n    tpm = test_preds.dot(fold_weights) / np.sum(fold_weights)\n    tpTuned = threshold_Rounder(tpm, thresholds)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n# **》》》Model3**\n---","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    thresholds = KappaOPtimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, thresholds)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, thresholds)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),    \n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Final Ensemble","metadata":{}},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\n# final_submission.to_csv('submission.csv', index=False)\n\nsub3.to_csv('submission.csv', index=False)\n\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}