{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Phát triển từ code gốc:\n\n- Giữ nguyên các giai đoạn xử lý dữ liệu, biểu đồ cùng với model tabnet\n- Thêm model SAINT (Self-Attention and Intersample Attention Transformer) thích hợp cho xử lý dữ liệu kiểu bảng\n- Ensemble với các mô hình khác như XGBoost, CatBoost, LightGBM, RandomForest, GradiantBoosting nhằm đưa ra kết quả cuối cùng\nNotebook gốc: https://www.kaggle.com/code/seonokrkim/lb0-494-with-tabnet","metadata":{}},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-12-19T13:58:52.208256Z","iopub.execute_input":"2024-12-19T13:58:52.208502Z","iopub.status.idle":"2024-12-19T13:59:33.646175Z","shell.execute_reply.started":"2024-12-19T13:58:52.208475Z","shell.execute_reply":"2024-12-19T13:59:33.645081Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nimport torch","metadata":{"execution":{"iopub.status.busy":"2024-12-19T13:59:33.648715Z","iopub.execute_input":"2024-12-19T13:59:33.649420Z","iopub.status.idle":"2024-12-19T13:59:38.127899Z","shell.execute_reply.started":"2024-12-19T13:59:33.649379Z","shell.execute_reply":"2024-12-19T13:59:38.126852Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"execution":{"iopub.status.busy":"2024-12-19T13:59:38.129090Z","iopub.execute_input":"2024-12-19T13:59:38.129458Z","iopub.status.idle":"2024-12-19T13:59:54.270873Z","shell.execute_reply.started":"2024-12-19T13:59:38.129431Z","shell.execute_reply":"2024-12-19T13:59:54.270209Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:54.271929Z","iopub.execute_input":"2024-12-19T13:59:54.272590Z","iopub.status.idle":"2024-12-19T13:59:54.354653Z","shell.execute_reply.started":"2024-12-19T13:59:54.272562Z","shell.execute_reply":"2024-12-19T13:59:54.353673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = []\n    for col in columns:\n        if data[col].dtype in ['object', 'category']:\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents = data[col].value_counts(normalize=True, dropna=False, sort=False) * 100\n            formatted = counts.astype(str) + ' (' + percents.round(2).astype(str) + '%)'\n            stats_col = pd.DataFrame({'count (%)': formatted})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().transpose()\n            stats_col['missing'] = data[col].isnull().sum()\n            stats_col.index.name = col\n            stats.append(stats_col)\n\n    return pd.concat(stats, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:54.357411Z","iopub.execute_input":"2024-12-19T13:59:54.357778Z","iopub.status.idle":"2024-12-19T13:59:54.364063Z","shell.execute_reply.started":"2024-12-19T13:59:54.357736Z","shell.execute_reply":"2024-12-19T13:59:54.363251Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's identify the features that are related to the target variable and that are not present in the test set.","metadata":{}},{"cell_type":"code","source":"train_cols = set(train.columns)\ntest_cols = set(test.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\ndata_dict[data_dict['Field'].isin(columns_not_in_test)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:54.365437Z","iopub.execute_input":"2024-12-19T13:59:54.365815Z","iopub.status.idle":"2024-12-19T13:59:54.401971Z","shell.execute_reply.started":"2024-12-19T13:59:54.365776Z","shell.execute_reply":"2024-12-19T13:59:54.401200Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Parent-Child Internet Addiction Test (PCIAT): contains 20 items (PCIAT-PCIAT_01 to PCIAT-PCIAT_20), each assessing a different aspect of a child's behavior related to internet use. The items are answered on a scale (from 0 to 5), and the total score provides an indication of the severity of internet addiction.\n\nWe also have season of participation in PCIAT-Season and total Score in PCIAT-PCIAT_Total; so there are 22 PCIAT test-related columns in total.\n\nLet's verify that the PCIAT-PCIAT_Total align with the corresponding sii categories by calculating its minimum and maximum scores for each sii category:","metadata":{}},{"cell_type":"code","source":"pciat_min_max = train.groupby('sii')['PCIAT-PCIAT_Total'].agg(['min', 'max'])\npciat_min_max = pciat_min_max.rename(\n    columns={'min': 'Minimum PCIAT total Score', 'max': 'Maximum total PCIAT Score'}\n)\npciat_min_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:54.402932Z","iopub.execute_input":"2024-12-19T13:59:54.403168Z","iopub.status.idle":"2024-12-19T13:59:54.420704Z","shell.execute_reply.started":"2024-12-19T13:59:54.403146Z","shell.execute_reply":"2024-12-19T13:59:54.419853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict[data_dict['Field'] == 'PCIAT-PCIAT_Total']['Value Labels'].iloc[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:54.421932Z","iopub.execute_input":"2024-12-19T13:59:54.422293Z","iopub.status.idle":"2024-12-19T13:59:54.436122Z","shell.execute_reply.started":"2024-12-19T13:59:54.422255Z","shell.execute_reply":"2024-12-19T13:59:54.435435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_labels = ['None', 'Mild', 'Moderate', 'Severe']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:54.437173Z","iopub.execute_input":"2024-12-19T13:59:54.437412Z","iopub.status.idle":"2024-12-19T13:59:54.446477Z","shell.execute_reply.started":"2024-12-19T13:59:54.437389Z","shell.execute_reply":"2024-12-19T13:59:54.445598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_dtype = pl.Enum(['Spring', 'Summer', 'Fall', 'Winter'])\n\ntrain = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n)\n\ntest = (\n    pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    .with_columns(pl.col('^.*Season$').cast(season_dtype))\n)\n\ntrain\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:54.447576Z","iopub.execute_input":"2024-12-19T13:59:54.447833Z","iopub.status.idle":"2024-12-19T13:59:54.619945Z","shell.execute_reply.started":"2024-12-19T13:59:54.447808Z","shell.execute_reply":"2024-12-19T13:59:54.619110Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The training dataset has 3960 samples (children who participate in the study) and 80 features (not counting the id column and the target sii). According to the documentation, the full test set comprises about 3800 instances, of which 1400 are public and 2400 private.","metadata":{}},{"cell_type":"markdown","source":"# **Missing values**\n\nAll columns have a substantial proportion of missing values, except id (not surprisingly) and the three basic demographic columns for sex, age and season of enrollment. Even the target sii has missing values:","metadata":{}},{"cell_type":"code","source":"missing_count = (\n    train\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .sort('null_count', descending=True)\n    .with_columns((pl.col('null_count') / len(train)).alias('null_ratio'))\n)\nplt.figure(figsize=(6, 15))\nplt.title('Missing values over the whole training dataset')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:54.620798Z","iopub.execute_input":"2024-12-19T13:59:54.621030Z","iopub.status.idle":"2024-12-19T13:59:55.643433Z","shell.execute_reply.started":"2024-12-19T13:59:54.621008Z","shell.execute_reply":"2024-12-19T13:59:55.642528Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"If we count the missing feature values only for the usable part of the training dataset (samples are usable for supervised training if the target is known), the chart looks slightly different:","metadata":{}},{"cell_type":"code","source":"supervised_usable = (\n    train\n    .filter(pl.col('sii').is_not_null())\n)\n\nmissing_count = (\n    supervised_usable\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .sort('null_count', descending=True)\n    .with_columns((pl.col('null_count') / len(supervised_usable)).alias('null_ratio'))\n)\nplt.figure(figsize=(6, 15))\nplt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:55.644419Z","iopub.execute_input":"2024-12-19T13:59:55.644652Z","iopub.status.idle":"2024-12-19T13:59:56.526430Z","shell.execute_reply.started":"2024-12-19T13:59:55.644628Z","shell.execute_reply":"2024-12-19T13:59:56.525565Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# The target\nThe target sii is available exactly for those participants for whom we have results of the Parent-Child Internet Addiction Test (PCIAT), and it is a function of the PCIAT total score:","metadata":{}},{"cell_type":"code","source":"print(train.select(pl.col('PCIAT-PCIAT_Total').is_null() == pl.col('sii').is_null()).to_series().mean())\n\n(train\n .select(pl.col('PCIAT-PCIAT_Total'))\n .group_by(train.get_column('sii'))\n .agg(pl.col('PCIAT-PCIAT_Total').min().alias('PCIAT-PCIAT_Total min'),\n      pl.col('PCIAT-PCIAT_Total').max().alias('PCIAT-PCIAT_Total max'),\n      pl.col('PCIAT-PCIAT_Total').len().alias('count'))\n .sort('sii')\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:56.527737Z","iopub.execute_input":"2024-12-19T13:59:56.528029Z","iopub.status.idle":"2024-12-19T13:59:56.561221Z","shell.execute_reply.started":"2024-12-19T13:59:56.528003Z","shell.execute_reply":"2024-12-19T13:59:56.560390Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Insight:\n\n- We should focus on predicting the target from all other features except the PCIAT results.\n- There are many missing values, we know the target only for two thirds of the samples.\n- This dataset is imbalanced. Half of the samples are in class 0, while very few in class 3.","metadata":{}},{"cell_type":"markdown","source":"# Demographics","metadata":{}},{"cell_type":"code","source":"_, axs = plt.subplots(2, 1, sharex=True)\nfor sex in range(2):\n    ax = axs.ravel()[sex]\n    vc = train.filter(pl.col('Basic_Demos-Sex') == sex).get_column('Basic_Demos-Age').value_counts()\n    ax.bar(vc.get_column('Basic_Demos-Age'),\n           vc.get_column('count'),\n           color=['lightblue', 'coral'][sex],\n           label=['boys', 'girls'][sex])\n    ax.xaxis.set_major_locator(MaxNLocator(integer=True))\n    ax.set_ylabel('count')\n    ax.legend()\nplt.suptitle('Age distribution')\naxs.ravel()[1].set_xlabel('years')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:56.564393Z","iopub.execute_input":"2024-12-19T13:59:56.564662Z","iopub.status.idle":"2024-12-19T13:59:56.951026Z","shell.execute_reply.started":"2024-12-19T13:59:56.564637Z","shell.execute_reply":"2024-12-19T13:59:56.950222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vc = train.get_column('Basic_Demos-Enroll_Season').value_counts()\nplt.pie(vc.get_column('count'), labels=vc.get_column('Basic_Demos-Enroll_Season'))\nplt.title('Season of enrollment')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:56.952508Z","iopub.execute_input":"2024-12-19T13:59:56.952869Z","iopub.status.idle":"2024-12-19T13:59:57.322938Z","shell.execute_reply.started":"2024-12-19T13:59:56.952831Z","shell.execute_reply":"2024-12-19T13:59:57.322056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The distribution of enrollment by season is relatively balanced, with the highest enrollment in Spring and the lowest in Fall","metadata":{}},{"cell_type":"code","source":"_, axs = plt.subplots(2, 1, sharex=True, sharey=True)\nfor sex in range(2):\n    ax = axs.ravel()[sex]\n    vc = train.filter(pl.col('Basic_Demos-Sex') == sex).get_column('sii').value_counts()\n    ax.bar(vc.get_column('sii'),\n           vc.get_column('count') / vc.get_column('count').sum(),\n           color=['lightblue', 'coral'][sex],\n           label=['boys', 'girls'][sex])\n    ax.set_xticks(np.arange(4), target_labels)\n    ax.yaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n    ax.set_ylabel('count')\n    ax.legend()\nplt.suptitle('Target distribution')\naxs.ravel()[1].set_xlabel('Severity Impairment Index (sii)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:57.324267Z","iopub.execute_input":"2024-12-19T13:59:57.324873Z","iopub.status.idle":"2024-12-19T13:59:57.564610Z","shell.execute_reply.started":"2024-12-19T13:59:57.324831Z","shell.execute_reply":"2024-12-19T13:59:57.563762Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Boys have a slightly higher risk of internet addiction than girls.","metadata":{}},{"cell_type":"markdown","source":"# Correlations","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(14, 12))\ncorr_matrix = supervised_usable.select([\n    'PCIAT-PCIAT_Total', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'Physical-BMI', \n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP', 'Physical-HeartRate',\n    'PreInt_EduHx-computerinternet_hoursday', 'SDS-SDS_Total_T', 'PAQ_A-PAQ_A_Total',\n    'PAQ_C-PAQ_C_Total', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec',\n    'FGC-FGC_CU', 'FGC-FGC_GSND','FGC-FGC_GSD','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL','BIA-BIA_Activity_Level_num', \n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI','BIA-BIA_FMI', 'BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n    'BIA-BIA_SMM','BIA-BIA_TBW'\n    # Add other relevant columns\n]).to_pandas().corr()\n\nsii_corr = corr_matrix['PCIAT-PCIAT_Total'].drop('PCIAT-PCIAT_Total')\nfiltered_corr = sii_corr[(sii_corr > 0.1) | (sii_corr < -0.1)]\n\nprint(filtered_corr)\n\nplt.figure(figsize=(8, 6))\nfiltered_corr.sort_values().plot(kind='barh', color='coral')\nplt.title('Features with Correlation > 0.1 or < -0.1 with PCIAT-PCIAT_Total')\nplt.xlabel('Correlation coefficient')\nplt.ylabel('Features')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:57.565891Z","iopub.execute_input":"2024-12-19T13:59:57.566670Z","iopub.status.idle":"2024-12-19T13:59:57.928154Z","shell.execute_reply.started":"2024-12-19T13:59:57.566628Z","shell.execute_reply":"2024-12-19T13:59:57.927303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Actigraphy files (time series)\n\nLooking at the file of participant id=0417c91e, a six-year old right-handed girl, we see that this participant started to use the accelerometer on a Tuesday (weekday=2) of the second quarter of the year at second 44100 of the day (12:15 PM), 5 days after the PCIAT test. She gave the accelerometer back on the 53rd day after the PCIAT test, a Monday of the third quarter, at 9:08 AM.","metadata":{}},{"cell_type":"code","source":"actigraphy = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nactigraphy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:57.929196Z","iopub.execute_input":"2024-12-19T13:59:57.929466Z","iopub.status.idle":"2024-12-19T13:59:58.018084Z","shell.execute_reply.started":"2024-12-19T13:59:57.929440Z","shell.execute_reply":"2024-12-19T13:59:58.017029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def analyze_actigraphy(id, only_one_week=False, small=False):\n    actigraphy = pl.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={id}/part-0.parquet')\n    day = actigraphy.get_column('relative_date_PCIAT') + actigraphy.get_column('time_of_day') / 86400e9\n    sample = train.filter(pl.col('id') == id)\n    age = sample.get_column('Basic_Demos-Age').item()\n    sex = ['boy', 'girl'][sample.get_column('Basic_Demos-Sex').item()]\n    actigraphy = (\n        actigraphy\n        .with_columns(\n            (day.diff() * 86400).alias('diff_seconds'),\n            (np.sqrt(np.square(pl.col('X')) + np.square(pl.col('Y')) + np.square(pl.col('Z'))).alias('norm'))\n        )\n    )\n\n    if only_one_week:\n        start = np.ceil(day.min())\n        mask = (start <= day.to_numpy()) & (day.to_numpy() <= start + 7*3)\n        mask &= ~ actigraphy.get_column('non-wear_flag').cast(bool).to_numpy()\n    else:\n        mask = np.full(len(day), True)\n        \n    if small:\n        timelines = [\n            ('enmo', 'forestgreen'),\n            ('light', 'orange'),\n        ]\n    else:\n        timelines = [\n            ('X', 'm'),\n            ('Y', 'm'),\n            ('Z', 'm'),\n#             ('norm', 'c'),\n            ('enmo', 'forestgreen'),\n            ('anglez', 'lightblue'),\n            ('light', 'orange'),\n            ('non-wear_flag', 'chocolate')\n    #         ('diff_seconds', 'k'),\n        ]\n        \n    _, axs = plt.subplots(len(timelines), 1, sharex=True, figsize=(12, len(timelines) * 1.1 + 0.5))\n    for ax, (feature, color) in zip(axs, timelines):\n        ax.set_facecolor('#eeeeee')\n        ax.scatter(day.to_numpy()[mask],\n                   actigraphy.get_column(feature).to_numpy()[mask],\n                   color=color, label=feature, s=1)\n        ax.legend(loc='upper left', facecolor='#eeeeee')\n        if feature == 'diff_seconds':\n            ax.set_ylim(-0.5, 20.5)\n    axs[-1].set_xlabel('day')\n    axs[-1].xaxis.set_major_locator(MaxNLocator(integer=True))\n    plt.tight_layout()\n    axs[0].set_title(f'id={id}, {sex}, age={age}')\n    plt.show()\n\nanalyze_actigraphy('0417c91e', only_one_week=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T13:59:58.019428Z","iopub.execute_input":"2024-12-19T13:59:58.019711Z","iopub.status.idle":"2024-12-19T14:00:00.087278Z","shell.execute_reply.started":"2024-12-19T13:59:58.019683Z","shell.execute_reply":"2024-12-19T14:00:00.086383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- We clearly see a daily pattern.\n- We see that the girl wore the accelerometer for 31 days and then took it off.\n- The dataset has a non-wear_flag column, but that flag is always zero for this participant.\n- The girl is in an environment where the illuminance exceeds 2500 lux every day (the device cannot measure more than 2500 lux). Such a high illuminance means that she is outdoors or in a room with huge windows.\n- The girl moves a lot: she has enmo values above 2 almost every day.\n- The time series usually contain measurements every 5 seconds, but some time steps are missing. It is not documented under what conditions time steps are skipped.","metadata":{}},{"cell_type":"markdown","source":"# Train Model","metadata":{}},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:00:00.088357Z","iopub.execute_input":"2024-12-19T14:00:00.088630Z","iopub.status.idle":"2024-12-19T14:00:00.092683Z","shell.execute_reply.started":"2024-12-19T14:00:00.088603Z","shell.execute_reply":"2024-12-19T14:00:00.091767Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:00:00.094194Z","iopub.execute_input":"2024-12-19T14:00:00.094442Z","iopub.status.idle":"2024-12-19T14:00:00.110582Z","shell.execute_reply.started":"2024-12-19T14:00:00.094418Z","shell.execute_reply":"2024-12-19T14:00:00.109780Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:00:00.111665Z","iopub.execute_input":"2024-12-19T14:00:00.111910Z","iopub.status.idle":"2024-12-19T14:01:35.896803Z","shell.execute_reply.started":"2024-12-19T14:00:00.111886Z","shell.execute_reply":"2024-12-19T14:01:35.895829Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:01:35.898492Z","iopub.execute_input":"2024-12-19T14:01:35.899367Z","iopub.status.idle":"2024-12-19T14:01:35.909120Z","shell.execute_reply.started":"2024-12-19T14:01:35.899310Z","shell.execute_reply":"2024-12-19T14:01:35.908155Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:01:35.910420Z","iopub.execute_input":"2024-12-19T14:01:35.910787Z","iopub.status.idle":"2024-12-19T14:01:35.923851Z","shell.execute_reply.started":"2024-12-19T14:01:35.910747Z","shell.execute_reply":"2024-12-19T14:01:35.922937Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  \n    'lambda_l2': 0.01, \n    'device': 'gpu'\n\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1, \n    'reg_lambda': 5, \n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10, \n    'task_type': 'GPU'\n\n}","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:01:35.924903Z","iopub.execute_input":"2024-12-19T14:01:35.925552Z","iopub.status.idle":"2024-12-19T14:01:35.942834Z","shell.execute_reply.started":"2024-12-19T14:01:35.925523Z","shell.execute_reply":"2024-12-19T14:01:35.941954Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport numpy as np\n\nclass SAINTWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, dim=32, depth=6, heads=8, dropout=0.1, ff_dropout=0.1, max_epochs=10, lr=1e-3):\n        self.dim = dim\n        self.depth = depth\n        self.heads = heads\n        self.dropout = dropout\n        self.ff_dropout = ff_dropout\n        self.max_epochs = max_epochs\n        self.lr = lr\n        self.imputer = SimpleImputer(strategy='median')\n        \n    def fit(self, X, y):\n        if hasattr(y, 'values'):\n            y = y.values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        X_tensor = torch.tensor(X_imputed, dtype=torch.float32)\n        y_tensor = torch.tensor(y, dtype=torch.long)\n        \n        X_cat = torch.zeros((X_imputed.shape[0], 1), dtype=torch.long)\n        \n        self.model = SAINT(\n            categories=[1], \n            num_continuous=X_imputed.shape[1],\n            dim=self.dim,\n            depth=self.depth,\n            heads=self.heads,\n            dim_out=4,\n            dropout=self.dropout,\n            ff_dropout=self.ff_dropout\n        )\n        \n        criterion = nn.CrossEntropyLoss()\n        optimizer = optim.Adam(self.model.parameters(), lr=self.lr)\n        \n        dataset = torch.utils.data.TensorDataset(X_cat, X_tensor, y_tensor)\n        loader = torch.utils.data.DataLoader(dataset, batch_size=64, shuffle=True)\n        \n        self.model.train()\n        for epoch in range(self.max_epochs):\n            for cat, cont, labels in loader:\n                optimizer.zero_grad()\n                \n                outputs = self.model(cat, cont)\n\n                loss = criterion(outputs, labels)\n\n                loss.backward()\n                optimizer.step()\n                \n        return self\n    \n    def predict(self, X):\n\n        self.model.eval()\n\n        X_imputed = self.imputer.transform(X)\n        X_tensor = torch.tensor(X_imputed, dtype=torch.float32)\n        X_cat = torch.zeros((X_imputed.shape[0], 1), dtype=torch.long)\n\n        with torch.no_grad():\n            outputs = self.model(X_cat, X_tensor)\n            probabilities = torch.softmax(outputs, dim=1)\n            \n            weights = torch.tensor([0, 1, 2, 3], dtype=torch.float32)\n            predictions = (probabilities * weights.unsqueeze(0)).sum(dim=1)\n            \n        return predictions.numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T14:01:35.944084Z","iopub.execute_input":"2024-12-19T14:01:35.944429Z","iopub.status.idle":"2024-12-19T14:01:35.964225Z","shell.execute_reply.started":"2024-12-19T14:01:35.944393Z","shell.execute_reply":"2024-12-19T14:01:35.963421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass SAINT(nn.Module):\n    def __init__(self, categories, num_continuous, dim, depth, heads, dim_out=1, dropout=0.1, ff_dropout=0.1):\n        super().__init__()\n        \n        self.cat_embeds = nn.ModuleList([nn.Embedding(c, dim) for c in categories])\n        \n        self.cont_embed = nn.Linear(num_continuous, dim)\n\n        self.transformer = nn.ModuleList([])\n        for _ in range(depth):\n            self.transformer.append(nn.ModuleList([\n                nn.LayerNorm(dim),\n                nn.MultiheadAttention(dim, heads, dropout=dropout),\n                nn.LayerNorm(dim),\n                nn.Sequential(\n                    nn.Linear(dim, dim * 4),\n                    nn.ReLU(),\n                    nn.Dropout(ff_dropout),\n                    nn.Linear(dim * 4, dim),\n                    nn.Dropout(ff_dropout)\n                )\n            ]))\n\n        self.norm = nn.LayerNorm(dim)\n        self.fc = nn.Linear(dim, dim_out)\n        \n    def forward(self, x_cat, x_cont):\n        if len(self.cat_embeds) > 0:\n            x_cat = [embed(x_cat[:, i]) for i, embed in enumerate(self.cat_embeds)]\n            x_cat = torch.stack(x_cat, dim=1)\n        else:\n            x_cat = torch.zeros((x_cont.size(0), 1, self.dim), device=x_cont.device)\n        \n        x_cont = self.cont_embed(x_cont).unsqueeze(1)\n\n        x = torch.cat((x_cat, x_cont), dim=1)\n\n        for norm1, attn, norm2, ff in self.transformer:\n            x = norm1(x)\n            x = x + attn(x, x, x)[0]\n            x = norm2(x)\n            x = x + ff(x)\n\n        x = self.norm(x)\n        x = x.mean(dim=1)\n        x = self.fc(x)\n        \n        return x\n\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        # Create internal validation set\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, \n            y, \n            test_size=0.2,\n            random_state=42\n        )\n        \n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=500,\n            patience=50,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        # Add deepcopy support for scikit-learn\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# TabNet hyperparameters\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Check if current metric is better than best\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)  # Save the entire model","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:01:35.965554Z","iopub.execute_input":"2024-12-19T14:01:35.965911Z","iopub.status.idle":"2024-12-19T14:01:35.995810Z","shell.execute_reply.started":"2024-12-19T14:01:35.965873Z","shell.execute_reply":"2024-12-19T14:01:35.994931Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) ","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:01:35.996897Z","iopub.execute_input":"2024-12-19T14:01:35.997243Z","iopub.status.idle":"2024-12-19T14:01:36.020711Z","shell.execute_reply.started":"2024-12-19T14:01:35.997201Z","shell.execute_reply":"2024-12-19T14:01:36.019889Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n])\n\nSubmission1 = TrainML(voting_model, test)\n\nSubmission1","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:01:36.021902Z","iopub.execute_input":"2024-12-19T14:01:36.022619Z","iopub.status.idle":"2024-12-19T14:04:29.658399Z","shell.execute_reply.started":"2024-12-19T14:01:36.022582Z","shell.execute_reply":"2024-12-19T14:04:29.657387Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)  \n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n}\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\nimport itertools\nfrom tqdm import tqdm\n\nparam_combinations = {\n    'dim': [64],\n    'depth': [8],\n    'heads': [4],\n    'dropout': [0.1],  \n    'ff_dropout': [0.1],\n    'max_epochs': [20],\n    'lr': [1e-3]\n}\n\nkeys = param_combinations.keys()\nvalues = param_combinations.values()\nall_combinations = [dict(zip(keys, v)) for v in itertools.product(*values)]\n\nresults = []\n\nfor params in tqdm(all_combinations, desc=\"Testing SAINT parameters\"):\n    try:\n        Light = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\n        XGB_Model = XGBRegressor(**XGB_Params)\n        CatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n        SAINT_Model = SAINTWrapper(**params)  \n        TabNet_Model = TabNetWrapper(**TabNet_Params) # New\n\n        ensemble = VotingRegressor(estimators=[\n            ('lightgbm', Light),\n            ('xgboost', XGB_Model),\n            ('catboost', CatBoost_Model),\n            ('saint', SAINT_Model),\n            ('tabnet', TabNet_Model)\n        ])\n\n        Submission_final, tKappa = TrainML(ensemble, test)\n\n        results.append({\n            **params,\n            'tKappa': tKappa\n        })\n        \n        print(f\"\\nParameters: {params}\")\n        print(f\"tKappa Score: {tKappa}\")\n        \n    except Exception as e:\n        print(f\"Error with params {params}: {str(e)}\")\n        continue\n\nresults_df = pd.DataFrame(results)\nresults_df = results_df.sort_values('tKappa', ascending=False)\n\nprint(\"\\nBest Parameters:\")\nprint(results_df.head())\n\nresults_df.to_csv('saint_parameter_search_results.csv', index=False) \n\nbest_params = results_df.iloc[0].drop('tKappa').to_dict()\nprint(\"\\nTraining final model with best parameters:\", best_params)\n\n\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nSAINT_Model = SAINTWrapper(**best_params) \nTabNet_Model = TabNetWrapper(**TabNet_Params) # New\n\nensemble = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('saint', SAINT_Model),\n    ('tabnet', TabNet_Model)\n], weights=[0.23, 0.23, 0.23, 0.08, 0.23])\n\n\nSubmission2 = TrainML(ensemble, test)\nSubmission2","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:04:29.659930Z","iopub.execute_input":"2024-12-19T14:04:29.660292Z","iopub.status.idle":"2024-12-19T14:14:01.033603Z","shell.execute_reply.started":"2024-12-19T14:04:29.660257Z","shell.execute_reply":"2024-12-19T14:14:01.032671Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))])),\n    ('saint', Pipeline(steps=[('imputer', imputer), ('regressor', SAINTWrapper(**best_params))])) \n])\n\nSubmission3 = TrainML(ensemble, test)\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:14:01.034852Z","iopub.execute_input":"2024-12-19T14:14:01.035149Z","iopub.status.idle":"2024-12-19T14:22:41.149493Z","shell.execute_reply.started":"2024-12-19T14:14:01.035121Z","shell.execute_reply":"2024-12-19T14:22:41.148599Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:22:41.150615Z","iopub.execute_input":"2024-12-19T14:22:41.150882Z","iopub.status.idle":"2024-12-19T14:22:41.164531Z","shell.execute_reply.started":"2024-12-19T14:22:41.150856Z","shell.execute_reply":"2024-12-19T14:22:41.163655Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"execution":{"iopub.status.busy":"2024-12-19T14:22:41.165702Z","iopub.execute_input":"2024-12-19T14:22:41.166103Z","iopub.status.idle":"2024-12-19T14:22:41.181370Z","shell.execute_reply.started":"2024-12-19T14:22:41.166061Z","shell.execute_reply":"2024-12-19T14:22:41.180644Z"},"trusted":true},"outputs":[],"execution_count":null}]}