{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"ℹ️ Info¶¶\n\n* **forked original great work kernels**\n    * https://www.kaggle.com/code/ichigoe/lb0-494-with-tabnet\n\n* **My upd(2024/11/08)**\n    * As a discussion issue for the original kernel.There was a problem that reproducibility was not guaranteed and LB was unstable.\n    * Therefore, Seed fixes such as Torch and LGB have been modified to ensure reproducibility.\n    * Below is the result of measuring LB with this kernel.\n        * [Version1.LB]0.492\n        * [Version2.LB]0.492\n        * [Version3.LB]0.492\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"---\n---","metadata":{}},{"cell_type":"markdown","source":"# Based\n- https://www.kaggle.com/code/honganzhu/cmi-piu-competition?scriptVersionId=201912528 Version44 LB0.492","metadata":{}},{"cell_type":"markdown","source":"》# Description of Imported Libraries\n\n- **NumPy (`np`)**: Used for efficient numerical operations, including linear algebra and array manipulation.\n- **Pandas (`pd`)**: Provides data structures like DataFrames for handling structured data, essential for data preprocessing.\n- **Polars (`pl`)**: A faster alternative to pandas for DataFrame operations, particularly useful for large datasets.\n- **Matplotlib & Seaborn (`plt`, `sns`)**: Visualization libraries. Matplotlib is used for basic plots, while Seaborn builds on it to create more advanced statistical visualizations.\n- **LightGBM, XGBoost, CatBoost**: Machine learning libraries used for gradient boosting, which is efficient for both regression and classification tasks.\n- **Colorama**: Enhances console output with colored text, making it easier to highlight important results or warnings.\n- **SciPy (`minimize`)**: Provides optimization routines, such as adjusting thresholds to maximize performance metrics like kappa scores.\n- **OS**: Used for file path manipulations and system-related functions.\n- **Scikit-learn (`sklearn`)**: A powerful machine learning library, providing utilities for cross-validation, metrics, and model cloning.\n- **YDF**: A specialized library for machine learning tasks, likely including decision forests.\n- **ThreadPoolExecutor & TQDM**: Tools for parallelizing tasks and displaying progress bars for long-running loops, improving efficiency and usability.\n- **Warnings**: Filters out unwanted warnings to keep the output clean, useful when dealing with noisy outputs from multiple libraries.\n- **IPython display (`clear_output`)**: A utility for clearing the Jupyter notebook output, often used to avoid clutter in long-running scripts.\n","metadata":{}},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:11.596395Z","iopub.execute_input":"2024-11-12T14:52:11.596798Z","iopub.status.idle":"2024-11-12T14:52:43.436195Z","shell.execute_reply.started":"2024-11-12T14:52:11.596749Z","shell.execute_reply":"2024-11-12T14:52:43.435078Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:43.438252Z","iopub.execute_input":"2024-11-12T14:52:43.438594Z","iopub.status.idle":"2024-11-12T14:52:46.025956Z","shell.execute_reply.started":"2024-11-12T14:52:43.438557Z","shell.execute_reply":"2024-11-12T14:52:46.024900Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:46.027363Z","iopub.execute_input":"2024-11-12T14:52:46.027890Z","iopub.status.idle":"2024-11-12T14:52:51.228649Z","shell.execute_reply.started":"2024-11-12T14:52:46.027830Z","shell.execute_reply":"2024-11-12T14:52:51.227850Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:52:51.231429Z","iopub.execute_input":"2024-11-12T14:52:51.232321Z","iopub.status.idle":"2024-11-12T14:52:51.240360Z","shell.execute_reply.started":"2024-11-12T14:52:51.232275Z","shell.execute_reply":"2024-11-12T14:52:51.239470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_labels = ['None', 'Mild', 'Moderate', 'Severe']","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:51.241551Z","iopub.execute_input":"2024-11-12T14:52:51.241966Z","iopub.status.idle":"2024-11-12T14:52:51.249682Z","shell.execute_reply.started":"2024-11-12T14:52:51.241919Z","shell.execute_reply":"2024-11-12T14:52:51.248768Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_dtype = pl.Enum(['Spring', 'Summer', 'Fall', 'Winter'])\n\ntrain = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n)\n\ntest = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n)\n\ntrain\ntest","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:51.250813Z","iopub.execute_input":"2024-11-12T14:52:51.251214Z","iopub.status.idle":"2024-11-12T14:52:51.294234Z","shell.execute_reply.started":"2024-11-12T14:52:51.251169Z","shell.execute_reply":"2024-11-12T14:52:51.293331Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"For a supervised learning, we need the target value, but some (sii) are missing. So we only use the part with valid target value(sii).","metadata":{}},{"cell_type":"code","source":"supervised_usable = (\n    train\n    .filter(pl.col('sii').is_not_null())\n)\n\nmissing_count = (\n    supervised_usable\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .sort('null_count', descending=True)\n    .with_columns((pl.col('null_count') / len(supervised_usable)).alias('null_ratio'))\n)\nplt.figure(figsize=(6, 15))\nplt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:51.295372Z","iopub.execute_input":"2024-11-12T14:52:51.295669Z","iopub.status.idle":"2024-11-12T14:52:52.538314Z","shell.execute_reply.started":"2024-11-12T14:52:51.295634Z","shell.execute_reply":"2024-11-12T14:52:52.537358Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.select(pl.col('PCIAT-PCIAT_Total').is_null() == pl.col('sii').is_null()).to_series().mean())\n\n(train\n .select(pl.col('PCIAT-PCIAT_Total'))\n .group_by(train.get_column('sii'))\n .agg(pl.col('PCIAT-PCIAT_Total').min().alias('PCIAT-PCIAT_Total min'),\n      pl.col('PCIAT-PCIAT_Total').max().alias('PCIAT-PCIAT_Total max'),\n      pl.col('PCIAT-PCIAT_Total').len().alias('count'))\n .sort('sii')\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:52.539532Z","iopub.execute_input":"2024-11-12T14:52:52.539850Z","iopub.status.idle":"2024-11-12T14:52:52.551832Z","shell.execute_reply.started":"2024-11-12T14:52:52.539815Z","shell.execute_reply":"2024-11-12T14:52:52.550781Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Insight:**\n\nThis dataset is imbalanced. Half of the samples are in class 0, while very few in class 3.\n","metadata":{}},{"cell_type":"code","source":"print('Columns missing in test:')\nprint([f for f in train.columns if f not in test.columns])","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:52.552906Z","iopub.execute_input":"2024-11-12T14:52:52.553217Z","iopub.status.idle":"2024-11-12T14:52:52.558356Z","shell.execute_reply.started":"2024-11-12T14:52:52.553184Z","shell.execute_reply":"2024-11-12T14:52:52.557387Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Demographics\n","metadata":{}},{"cell_type":"markdown","source":"Now we look at some basic demographics.","metadata":{}},{"cell_type":"code","source":"vc = train.get_column('Basic_Demos-Enroll_Season').value_counts()\nplt.pie(vc.get_column('count'), labels=vc.get_column('Basic_Demos-Enroll_Season'))\nplt.title('Season of enrollment')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:52.562582Z","iopub.execute_input":"2024-11-12T14:52:52.562940Z","iopub.status.idle":"2024-11-12T14:52:52.694547Z","shell.execute_reply.started":"2024-11-12T14:52:52.562858Z","shell.execute_reply":"2024-11-12T14:52:52.691585Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vc = train.get_column('Basic_Demos-Sex').value_counts()\nplt.pie(vc.get_column('count'), labels=['boys', 'girls'])\nplt.title('Sex of participant')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:52.696978Z","iopub.execute_input":"2024-11-12T14:52:52.697928Z","iopub.status.idle":"2024-11-12T14:52:52.847425Z","shell.execute_reply.started":"2024-11-12T14:52:52.697835Z","shell.execute_reply":"2024-11-12T14:52:52.845755Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_, axs = plt.subplots(2, 1, sharex=True)\nfor sex in range(2):\n    ax = axs.ravel()[sex]\n    vc = train.filter(pl.col('Basic_Demos-Sex') == sex).get_column('Basic_Demos-Age').value_counts()\n    ax.bar(vc.get_column('Basic_Demos-Age'),\n           vc.get_column('count'),\n           color=['lightblue', 'coral'][sex],\n           label=['boys', 'girls'][sex])\n    ax.xaxis.set_major_locator(MaxNLocator(integer=True))\n    ax.set_ylabel('count')\n    ax.legend()\nplt.suptitle('Age distribution')\naxs.ravel()[1].set_xlabel('years')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:52.849535Z","iopub.execute_input":"2024-11-12T14:52:52.850208Z","iopub.status.idle":"2024-11-12T14:52:53.353476Z","shell.execute_reply.started":"2024-11-12T14:52:52.850144Z","shell.execute_reply":"2024-11-12T14:52:53.352581Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_, axs = plt.subplots(2, 1, sharex=True, sharey=True)\nfor sex in range(2):\n    ax = axs.ravel()[sex]\n    vc = train.filter(pl.col('Basic_Demos-Sex') == sex).get_column('sii').value_counts()\n    ax.bar(vc.get_column('sii'),\n           vc.get_column('count') / vc.get_column('count').sum(),\n           color=['lightblue', 'coral'][sex],\n           label=['boys', 'girls'][sex])\n    ax.set_xticks(np.arange(4), target_labels)\n    ax.yaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n    ax.set_ylabel('count')\n    ax.legend()\nplt.suptitle('Target distribution')\naxs.ravel()[1].set_xlabel('Severity Impairment Index (sii)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:53.354748Z","iopub.execute_input":"2024-11-12T14:52:53.355076Z","iopub.status.idle":"2024-11-12T14:52:53.706900Z","shell.execute_reply.started":"2024-11-12T14:52:53.355041Z","shell.execute_reply":"2024-11-12T14:52:53.705936Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Now we look at correlations","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(14, 12))\ncorr_matrix = supervised_usable.select([\n    'PCIAT-PCIAT_Total', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'Physical-BMI', \n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP', 'Physical-HeartRate',\n    'PreInt_EduHx-computerinternet_hoursday', 'SDS-SDS_Total_T', 'PAQ_A-PAQ_A_Total',\n    'PAQ_C-PAQ_C_Total', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec',\n    'FGC-FGC_CU', 'FGC-FGC_GSND','FGC-FGC_GSD','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL','BIA-BIA_Activity_Level_num', \n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI','BIA-BIA_FMI', 'BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n    'BIA-BIA_SMM','BIA-BIA_TBW'\n    # Add other relevant columns\n]).to_pandas().corr()\n\nsii_corr = corr_matrix['PCIAT-PCIAT_Total'].drop('PCIAT-PCIAT_Total')\nfiltered_corr = sii_corr[(sii_corr > 0.1) | (sii_corr < -0.1)]\n\nprint(filtered_corr)\n\nplt.figure(figsize=(8, 6))\nfiltered_corr.sort_values().plot(kind='barh', color='coral')\nplt.title('Features with Correlation > 0.1 or < -0.1 with PCIAT-PCIAT_Total')\nplt.xlabel('Correlation coefficient')\nplt.ylabel('Features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:53.708155Z","iopub.execute_input":"2024-11-12T14:52:53.708471Z","iopub.status.idle":"2024-11-12T14:52:54.116956Z","shell.execute_reply.started":"2024-11-12T14:52:53.708437Z","shell.execute_reply":"2024-11-12T14:52:54.115957Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Actigraphy (time series)","metadata":{}},{"cell_type":"code","source":"actigraphy = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nactigraphy","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:54.118400Z","iopub.execute_input":"2024-11-12T14:52:54.119094Z","iopub.status.idle":"2024-11-12T14:52:54.145112Z","shell.execute_reply.started":"2024-11-12T14:52:54.119043Z","shell.execute_reply":"2024-11-12T14:52:54.144160Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def analyze_actigraphy(id, only_one_week=False, small=False):\n    actigraphy = pl.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={id}/part-0.parquet')\n    day = actigraphy.get_column('relative_date_PCIAT') + actigraphy.get_column('time_of_day') / 86400e9\n    sample = train.filter(pl.col('id') == id)\n    age = sample.get_column('Basic_Demos-Age').item()\n    sex = ['boy', 'girl'][sample.get_column('Basic_Demos-Sex').item()]\n    actigraphy = (\n        actigraphy\n        .with_columns(\n            (day.diff() * 86400).alias('diff_seconds'),\n            (np.sqrt(np.square(pl.col('X')) + np.square(pl.col('Y')) + np.square(pl.col('Z'))).alias('norm'))\n        )\n    )\n\n    if only_one_week:\n        start = np.ceil(day.min())\n        mask = (start <= day.to_numpy()) & (day.to_numpy() <= start + 7*3)\n        mask &= ~ actigraphy.get_column('non-wear_flag').cast(bool).to_numpy()\n    else:\n        mask = np.full(len(day), True)\n        \n    if small:\n        timelines = [\n            ('enmo', 'forestgreen'),\n            ('light', 'orange'),\n        ]\n    else:\n        timelines = [\n            ('X', 'm'),\n            ('Y', 'm'),\n            ('Z', 'm'),\n#             ('norm', 'c'),\n            ('enmo', 'forestgreen'),\n            ('anglez', 'lightblue'),\n            ('light', 'orange'),\n            ('non-wear_flag', 'chocolate')\n    #         ('diff_seconds', 'k'),\n        ]\n        \n    _, axs = plt.subplots(len(timelines), 1, sharex=True, figsize=(12, len(timelines) * 1.1 + 0.5))\n    for ax, (feature, color) in zip(axs, timelines):\n        ax.set_facecolor('#eeeeee')\n        ax.scatter(day.to_numpy()[mask],\n                   actigraphy.get_column(feature).to_numpy()[mask],\n                   color=color, label=feature, s=1)\n        ax.legend(loc='upper left', facecolor='#eeeeee')\n        if feature == 'diff_seconds':\n            ax.set_ylim(-0.5, 20.5)\n    axs[-1].set_xlabel('day')\n    axs[-1].xaxis.set_major_locator(MaxNLocator(integer=True))\n    plt.tight_layout()\n    axs[0].set_title(f'id={id}, {sex}, age={age}')\n    plt.show()\n\nanalyze_actigraphy('0417c91e', only_one_week=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:54.146422Z","iopub.execute_input":"2024-11-12T14:52:54.146831Z","iopub.status.idle":"2024-11-12T14:52:56.470797Z","shell.execute_reply.started":"2024-11-12T14:52:54.146784Z","shell.execute_reply":"2024-11-12T14:52:56.469862Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:56.472315Z","iopub.execute_input":"2024-11-12T14:52:56.473084Z","iopub.status.idle":"2024-11-12T14:52:56.477419Z","shell.execute_reply.started":"2024-11-12T14:52:56.473028Z","shell.execute_reply":"2024-11-12T14:52:56.476402Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering\n\n- **Feature Selection**: The dataset contains features related to physical characteristics (e.g., BMI, Height, Weight), behavioral aspects (e.g., internet usage), and fitness data (e.g., endurance time). \n- **Categorical Feature Encoding**: Categorical features are mapped to numerical values using custom mappings for each unique category within the dataset. This ensures compatibility with machine learning algorithms that require numerical input.\n- **Time Series Aggregation**: Time series statistics (e.g., mean, standard deviation) from the actigraphy data are computed and merged into the main dataset to create additional features for model training.\n","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom scipy.stats import linregress\n\ndef transform_time_of_day(df):\n    \"\"\"Convert `time_of_day` to hours within a day.\"\"\"\n    df['hour'] = (df['time_of_day'] % (24 * 60 * 60 * 1000)) / (60 * 60 * 1000)  # Convert ms to hours\n    return df\n\ndef impute_missing_values(features):\n    \"\"\"Impute missing values in features with appropriate strategies.\"\"\"\n    # Fill NaNs for mean features with the mean of the feature\n    for col in features:\n        if \"mean\" in col or \"avg\" in col:\n            features[col].fillna(features[col].mean(), inplace=True)\n        elif \"std\" in col or \"variation\" in col:\n            # Impute std or variation with 0 as they denote no variation when missing\n            features[col].fillna(0, inplace=True)\n        elif \"spike_count\" in col or \"count\" in col:\n            # Impute spike counts with 0, assuming no spikes if missing\n            features[col].fillna(0, inplace=True)\n        else:\n            # Fill other NaNs with median\n            features[col].fillna(features[col].median(), inplace=True)\n    return features\n\ndef extract_time_series_features(actigraphy_df):\n    # Transform time_of_day to manageable format and add derived columns\n    actigraphy_df = transform_time_of_day(actigraphy_df)\n    \n    # Define time segments\n    night_hours = (0, 6)\n\n    # Initialize dictionary for features\n    features = {}\n\n    # ENMO Features\n    features['weekly_enmo_mean'] = actigraphy_df['enmo'].mean()\n    features['weekly_enmo_std'] = actigraphy_df['enmo'].std()\n    features['night_enmo_mean'] = actigraphy_df[(actigraphy_df['hour'] >= night_hours[0]) & (actigraphy_df['hour'] < night_hours[1])]['enmo'].mean()\n    \n    # Moving averages (2-hour and 6-hour)\n    actigraphy_df['enmo_ma_2h'] = actigraphy_df['enmo'].rolling(window=120, min_periods=1).mean()\n    actigraphy_df['enmo_ma_6h'] = actigraphy_df['enmo'].rolling(window=360, min_periods=1).mean()\n    features['enmo_2h_ma_avg'] = actigraphy_df['enmo_ma_2h'].mean()\n    features['enmo_6h_ma_avg'] = actigraphy_df['enmo_ma_6h'].mean()\n\n    # Gradients and Activity Spikes\n    features['enmo_gradient_mean'] = actigraphy_df['enmo'].diff().mean()\n    features['enmo_spike_count'] = (actigraphy_df['enmo'].diff() > 0.1).sum()  # Spike threshold at 0.1\n\n    # Seasonal (Quarterly) ENMO average and comparison to overall average\n    actigraphy_df['quarter'] = (actigraphy_df['time_of_day'] // (3 * 30 * 24 * 60 * 60 * 1000)) % 4 + 1\n    quarterly_enmo = actigraphy_df.groupby('quarter')['enmo'].mean()\n    features['enmo_q1_avg'] = quarterly_enmo.get(1, np.nan)\n    features['enmo_q2_avg'] = quarterly_enmo.get(2, np.nan)\n    features['enmo_q3_avg'] = quarterly_enmo.get(3, np.nan)\n    features['enmo_q4_avg'] = quarterly_enmo.get(4, np.nan)\n    features['enmo_seasonal_variation'] = quarterly_enmo.std()\n\n    # Light-based Features\n    night_light = actigraphy_df[(actigraphy_df['hour'] >= night_hours[0]) & (actigraphy_df['hour'] < night_hours[1])]\n    features['avg_night_light'] = night_light['light'].mean()\n    features['night_light_spike_count'] = (night_light['light'].diff() > 0.2).sum()  # Spike threshold at 0.2\n    \n    # Battery-based Features\n    battery_drop_count = (actigraphy_df['battery_voltage'].diff() < 0).sum()\n    features['battery_drop_count'] = battery_drop_count\n\n    # Non-Wear Percentage\n    features['non_wear_percentage'] = actigraphy_df['non-wear_flag'].mean()\n    \n    # AngleZ-based Posture Deterioration Detection\n    actigraphy_df['anglez_grad'] = actigraphy_df['anglez'].diff()\n    features['avg_anglez'] = actigraphy_df['anglez'].mean()\n    features['anglez_std'] = actigraphy_df['anglez'].std()\n    features['posture_deterioration_spikes'] = (actigraphy_df['anglez_grad'].abs() > 5).sum()  # Spike threshold at 5 degrees\n    \n    # Trend Analysis\n    trend_enmo = linregress(range(len(actigraphy_df)), actigraphy_df['enmo'])[0]  # Slope of ENMO trend\n    trend_light = linregress(range(len(actigraphy_df)), actigraphy_df['light'])[0]  # Slope of Light trend\n    features['enmo_trend_slope'] = trend_enmo\n    features['light_trend_slope'] = trend_light\n\n    # Temporal Features for Activity\n    features['nighttime_activity_proportion'] = night_light['enmo'].sum() / actigraphy_df['enmo'].sum()\n    \n    # Activity during weekday vs weekend\n    weekday_enmo = actigraphy_df[actigraphy_df['weekday'] < 5]['enmo'].mean()\n    weekend_enmo = actigraphy_df[actigraphy_df['weekday'] >= 5]['enmo'].mean()\n    features['weekday_enmo_avg'] = weekday_enmo\n    features['weekend_enmo_avg'] = weekend_enmo\n\n    # Convert features dictionary to DataFrame and Impute Missing Values\n    features_df = pd.DataFrame(features, index=[0])\n    features_df = impute_missing_values(features_df)\n    return features_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-12T14:52:56.478997Z","iopub.execute_input":"2024-11-12T14:52:56.479390Z","iopub.status.idle":"2024-11-12T14:52:56.503060Z","shell.execute_reply.started":"2024-11-12T14:52:56.479335Z","shell.execute_reply":"2024-11-12T14:52:56.502043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\n\ndef process_file(filename, dirname):\n    \"\"\"Process each file to extract features, concatenate with descriptive stats, and flatten the resulting DataFrame.\"\"\"\n    # Load the DataFrame from file\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    \n    # Extract features\n    features_df = extract_time_series_features(df)\n    \n    # Compute full descriptive statistics and flatten to a 1D feature vector\n    describe_df = df.describe().reset_index(drop=True)\n    describe_flattened = describe_df.values.flatten()\n    \n    # Add the descriptive statistics to the feature DataFrame\n    describe_features = pd.DataFrame([describe_flattened], columns=[f\"{col}_{stat}\" for col in df.columns for stat in describe_df.index])\n    combined_features_df = pd.concat([features_df, describe_features], axis=1)\n    \n    # Flatten combined features to a 1D array\n    features_flat = combined_features_df.values.flatten()\n    \n    # Return flattened features and id extracted from filename\n    return features_flat, filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    \"\"\"Load and process all files in the directory and compile into a single DataFrame.\"\"\"\n    ids = os.listdir(dirname)\n    \n    # Using ThreadPoolExecutor for parallel processing\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    # Unpack results into stats (flattened features) and indexes\n    stats, indexes = zip(*results)\n\n    # Convert stats into DataFrame with column names\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])  # Each row corresponds to a file's features\n    df['id'] = indexes  # Add id column\n    \n    return df\n\n# Example usage:\n# train_ts = load_time_series(\"/path/to/series_train.parquet\")\n\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:56.504594Z","iopub.execute_input":"2024-11-12T14:52:56.505314Z","iopub.status.idle":"2024-11-12T14:52:56.529501Z","shell.execute_reply.started":"2024-11-12T14:52:56.505265Z","shell.execute_reply":"2024-11-12T14:52:56.528538Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Function to scale data\ndef scale_data(df):\n    \"\"\"Scale the data using StandardScaler and return the scaled data and scaler.\"\"\"\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    return df_scaled, scaler\n\n# Load data\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# Prepare time series data and drop 'id' column for scaling\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n# Identify constant columns\nconstant_columns = [col for col in df_train.columns if df_train[col].nunique() == 1]\n\n# Drop constant columns from train and test sets\ndf_train = df_train.drop(columns=constant_columns)\ndf_test = df_test.drop(columns=constant_columns)\n\nsmall_variance_columns = df_train.columns[df_train.std() < 1e-5]  # Adjust the threshold as needed\nfor col in small_variance_columns:\n    df_train[col] = df_train[col].fillna(df_train[col].mean())\n    df_test[col] = df_test[col].fillna(df_test[col].mean())\n\n# Scale the training and testing data separately\ndf_train_scaled, train_scaler = scale_data(df_train)\ndf_test_scaled = train_scaler.transform(df_test)\n# Convert scaled data to DataFrames\ndf_train_scaled = pd.DataFrame(df_train_scaled, columns=df_train.columns)\ndf_test_scaled = pd.DataFrame(df_test_scaled, columns=df_test.columns)\ndf_train_scaled.fillna(0, inplace=True)\ndf_test_scaled.fillna(0, inplace=True)\n# Confirm no NaNs\nprint(\"NaNs in scaled train data:\", df_train_scaled.isna().sum().sum())\nprint(\"NaNs in scaled test data:\", df_test_scaled.isna().sum().sum())\n\n\n# Perform autoencoder encoding on the scaled data\ntrain_ts_encoded = perform_autoencoder(df_train_scaled, encoding_dim=45, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test_scaled, encoding_dim=45, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n# Add 'id' column back to encoded data for merging\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded[\"id\"] = test_ts[\"id\"]\n\n# Merge encoded time series features back with main train and test data\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\n# Proceed with imputation, feature engineering, and other processing steps as in your original code\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\n# Define and filter selected feature columns\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD',\n                'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', \n                'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', \n                'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', \n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', \n                'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', \n                'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday', 'sii', \n                'BMI_Age', 'Internet_Hours_Age', 'BMI_Internet_Hours', 'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', \n                'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight', 'SMM_Height', 'Muscle_to_Fat', \n                'Hydration_Status', 'ICW_TBW'] + time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ntest = test[[c for c in featuresCols if c !=\"sii\"]]\n","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:52:56.530763Z","iopub.execute_input":"2024-11-12T14:52:56.531104Z","iopub.status.idle":"2024-11-12T14:57:32.668699Z","shell.execute_reply.started":"2024-11-12T14:52:56.531047Z","shell.execute_reply":"2024-11-12T14:57:32.667589Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:57:32.670121Z","iopub.execute_input":"2024-11-12T14:57:32.670843Z","iopub.status.idle":"2024-11-12T14:57:32.682683Z","shell.execute_reply.started":"2024-11-12T14:57:32.670801Z","shell.execute_reply":"2024-11-12T14:57:32.681706Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Training and Evaluation\n\n- **Model Types**: Various models are used, including:\n  - **LightGBM**: A gradient-boosting framework known for its speed and efficiency with large datasets.\n  - **XGBoost**: Another powerful gradient-boosting model used for structured data.\n  - **CatBoost**: Optimized for categorical features without the need for extensive preprocessing.\n  - **Voting Regressor**: An ensemble model that combines the predictions of LightGBM, XGBoost, and CatBoost for better accuracy.\n- **Cross-Validation**: Stratified K-Folds cross-validation is employed to split the data into training and validation sets, ensuring balanced class distribution in each fold.\n- **Quadratic Weighted Kappa (QWK)**: The performance of the models is evaluated using QWK, which measures the agreement between predicted and actual values, taking into account the ordinal nature of the target variable.\n- **Threshold Optimization**: The `minimize` function from `scipy.optimize` is used to fine-tune decision thresholds that map continuous predictions to discrete categories (None, Mild, Moderate, Severe).\n","metadata":{}},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:57:32.684113Z","iopub.execute_input":"2024-11-12T14:57:32.684414Z","iopub.status.idle":"2024-11-12T14:57:32.697782Z","shell.execute_reply.started":"2024-11-12T14:57:32.684382Z","shell.execute_reply":"2024-11-12T14:57:32.696852Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# Hyperparameter Tuning\n\n- **LightGBM Parameters**: Hyperparameters such as `learning_rate`, `max_depth`, `num_leaves`, and `feature_fraction` are tuned to improve the performance of the LightGBM model. These parameters control the complexity of the model and its ability to generalize to new data.\n- **XGBoost and CatBoost Parameters**: Similar tuning is applied for XGBoost and CatBoost, adjusting parameters such as `n_estimators`, `max_depth`, `learning_rate`, `subsample`, and `regularization` terms (`reg_alpha`, `reg_lambda`). These help in controlling overfitting and ensuring the model's robustness.","metadata":{}},{"cell_type":"code","source":"# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'cpu'\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 20,  # Increase this value\n    'task_type': 'GPU'\n\n}","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:57:32.699077Z","iopub.execute_input":"2024-11-12T14:57:32.699458Z","iopub.status.idle":"2024-11-12T14:57:32.709231Z","shell.execute_reply.started":"2024-11-12T14:57:32.699412Z","shell.execute_reply":"2024-11-12T14:57:32.708335Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# New: TabNet\n\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        # Create internal validation set\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, \n            y, \n            test_size=0.2,\n            random_state=42\n        )\n        \n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=200,\n            patience=20,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        # Add deepcopy support for scikit-learn\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# TabNet hyperparameters\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Check if current metric is better than best\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)  # Save the entire model","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:57:32.710463Z","iopub.execute_input":"2024-11-12T14:57:32.710766Z","iopub.status.idle":"2024-11-12T14:57:32.734615Z","shell.execute_reply.started":"2024-11-12T14:57:32.710733Z","shell.execute_reply":"2024-11-12T14:57:32.733471Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensemble Learning and Submission Preparation\n\n- **Ensemble Learning**: The model uses a **Voting Regressor**, which combines the predictions from LightGBM, XGBoost, and CatBoost. This approach is beneficial as it leverages the strengths of multiple models, reducing overfitting and improving overall model performance.\n- **Out-of-Fold (OOF) Predictions**: During cross-validation, out-of-fold predictions are generated for the training set, which helps in model evaluation without data leakage.\n- **Kappa Optimizer**: The Kappa Optimizer ensures that the predicted values are as close to the actual values as possible by adjusting the thresholds used to convert raw model outputs into class labels.\n- **Test Set Predictions**: After the model is trained and thresholds are optimized, the test dataset is processed, and predictions are generated using the ensemble model. These predictions are converted into the appropriate format for submission.\n- **Submission File Creation**: The predictions are saved in a CSV file following the required format for submission (e.g., for a Kaggle competition), which includes columns like `id` and `sii` (Severity Impairment Index).","metadata":{}},{"cell_type":"markdown","source":"# Final Results and Performance Metrics\n\n- **Train and Validation Scores**: After training across multiple folds, the mean Quadratic Weighted Kappa (QWK) score is calculated for both the training and validation datasets, providing an indicator of model performance. \n- **Optimized QWK Score**: The final optimized QWK score after threshold tuning is displayed, showcasing the model's ability to predict the severity levels effectively.\n- **Test Predictions**: The test set predictions are evaluated, and a breakdown of the predicted severity levels (None, Mild, Moderate, Severe) is shown, along with their respective counts.","metadata":{}},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) # New","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:57:32.735755Z","iopub.execute_input":"2024-11-12T14:57:32.736192Z","iopub.status.idle":"2024-11-12T14:57:32.747036Z","shell.execute_reply.started":"2024-11-12T14:57:32.736137Z","shell.execute_reply":"2024-11-12T14:57:32.746280Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n# **》》》Model1.Train**\n---","metadata":{}},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n],weights=[4.0,4.0,5.0,4.0])\n\nSubmission1 = TrainML(voting_model, test)\n\nSubmission1","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:57:32.748189Z","iopub.execute_input":"2024-11-12T14:57:32.748548Z","iopub.status.idle":"2024-11-12T14:59:10.579551Z","shell.execute_reply.started":"2024-11-12T14:57:32.748513Z","shell.execute_reply":"2024-11-12T14:59:10.578556Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"```\n],weights=[5.0,4.0,4.0,4.0])\nMean Train QWK --> 0.7424\nMean Validation QWK ---> 0.4735\n----> || Optimized QWK SCORE ::  0.533\n\n```","metadata":{}},{"cell_type":"markdown","source":"---\n# **》》》Model2**\n---","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"execution":{"iopub.status.busy":"2024-11-12T14:59:10.581377Z","iopub.execute_input":"2024-11-12T14:59:10.581943Z","iopub.status.idle":"2024-11-12T15:01:24.097752Z","shell.execute_reply.started":"2024-11-12T14:59:10.581872Z","shell.execute_reply":"2024-11-12T15:01:24.096796Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"execution":{"iopub.status.busy":"2024-11-12T15:01:24.102595Z","iopub.execute_input":"2024-11-12T15:01:24.102914Z","iopub.status.idle":"2024-11-12T15:04:51.715016Z","shell.execute_reply.started":"2024-11-12T15:01:24.102868Z","shell.execute_reply":"2024-11-12T15:04:51.714086Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"execution":{"iopub.status.busy":"2024-11-12T15:04:51.716314Z","iopub.execute_input":"2024-11-12T15:04:51.717223Z","iopub.status.idle":"2024-11-12T15:04:51.735210Z","shell.execute_reply.started":"2024-11-12T15:04:51.717172Z","shell.execute_reply":"2024-11-12T15:04:51.734299Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"execution":{"iopub.status.busy":"2024-11-12T15:04:51.736475Z","iopub.execute_input":"2024-11-12T15:04:51.737149Z","iopub.status.idle":"2024-11-12T15:04:51.748865Z","shell.execute_reply.started":"2024-11-12T15:04:51.737101Z","shell.execute_reply":"2024-11-12T15:04:51.747943Z"},"trusted":true},"outputs":[],"execution_count":null}]}