{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport sklearn\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\npd.set_option('display.max_rows', None)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:08.381471Z","iopub.execute_input":"2024-12-21T03:13:08.381873Z","iopub.status.idle":"2024-12-21T03:13:09.633310Z","shell.execute_reply.started":"2024-12-21T03:13:08.381842Z","shell.execute_reply":"2024-12-21T03:13:09.632270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"working_dir = '/kaggle/input/child-mind-institute-problematic-internet-use/'\noutput_dir = '/kaggle/working/'\ntrain_data = pd.read_csv(working_dir + 'train.csv')\ntest_data = pd.read_csv(working_dir + 'test.csv')\ndata_dict = pd.read_csv(working_dir + 'data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:09.634640Z","iopub.execute_input":"2024-12-21T03:13:09.635095Z","iopub.status.idle":"2024-12-21T03:13:09.731464Z","shell.execute_reply.started":"2024-12-21T03:13:09.635062Z","shell.execute_reply":"2024-12-21T03:13:09.730520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train_data.drop(columns='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:09.732883Z","iopub.execute_input":"2024-12-21T03:13:09.733205Z","iopub.status.idle":"2024-12-21T03:13:09.748099Z","shell.execute_reply.started":"2024-12-21T03:13:09.733163Z","shell.execute_reply":"2024-12-21T03:13:09.747156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encoding string data\nfrom sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\nle.fit(['Spring', 'Summer', 'Fall', 'Winter', 'missing'])\n\nseason_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n       'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season',\n       'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\nfor col in season_cols:\n    train_data[col] = train_data[col].fillna('missing')\n    train_data[col] = le.transform(train_data[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:09.749658Z","iopub.execute_input":"2024-12-21T03:13:09.750070Z","iopub.status.idle":"2024-12-21T03:13:09.820623Z","shell.execute_reply.started":"2024-12-21T03:13:09.750034Z","shell.execute_reply":"2024-12-21T03:13:09.819595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Trimming data by SII and PCIAT\ntrain_data = train_data[train_data['sii'].isnull() == False]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:09.821869Z","iopub.execute_input":"2024-12-21T03:13:09.822260Z","iopub.status.idle":"2024-12-21T03:13:09.830033Z","shell.execute_reply.started":"2024-12-21T03:13:09.822221Z","shell.execute_reply":"2024-12-21T03:13:09.828941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_columns = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', \n                 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', \n                 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', \n                 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', \n                 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:09.830725Z","iopub.execute_input":"2024-12-21T03:13:09.831035Z","iopub.status.idle":"2024-12-21T03:13:09.837159Z","shell.execute_reply.started":"2024-12-21T03:13:09.831009Z","shell.execute_reply":"2024-12-21T03:13:09.836223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dropoping columns with False PCIAT results\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)\ntrain_data.loc[train_data['PCIAT-PCIAT_Total'] == 0]\n\ntrain_data = train_data.drop(index=train_data.loc[train_data['PCIAT-PCIAT_Total'] == 0].index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:11.555031Z","iopub.execute_input":"2024-12-21T03:13:11.555387Z","iopub.status.idle":"2024-12-21T03:13:11.566751Z","shell.execute_reply.started":"2024-12-21T03:13:11.555360Z","shell.execute_reply":"2024-12-21T03:13:11.565642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop rows with missing PCIAT\nfor col in pciat_columns:\n    train_data = train_data.drop(index=train_data.loc[train_data[col].isnull()].index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:12.180860Z","iopub.execute_input":"2024-12-21T03:13:12.181256Z","iopub.status.idle":"2024-12-21T03:13:12.223848Z","shell.execute_reply.started":"2024-12-21T03:13:12.181215Z","shell.execute_reply":"2024-12-21T03:13:12.222805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop data with high missing ratio\nfor col in train_data.columns:\n    percent_missing = np.round(train_data[col].isnull().sum() / len(train_data) * 100, 2)\n    print(f\"{col.ljust(30)} {str(percent_missing).rjust(6)}%\")","metadata":{"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:12.375028Z","iopub.execute_input":"2024-12-21T03:13:12.375386Z","iopub.status.idle":"2024-12-21T03:13:12.418018Z","shell.execute_reply.started":"2024-12-21T03:13:12.375357Z","shell.execute_reply":"2024-12-21T03:13:12.417028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"high_miss_ratio = ['Physical-Waist_Circumference', 'Fitness_Endurance-Season',\n                   'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n                   'Fitness_Endurance-Time_Sec', 'FGC-FGC_GSND',\n                   'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone']\n# Drop PAQ_C-PAQ_C_Total because this is the score for children and the PAQ_A-PAQ_A_Total is for adolescent which was removed\n\ntrain_data.drop(columns=high_miss_ratio, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:16.277589Z","iopub.execute_input":"2024-12-21T03:13:16.278007Z","iopub.status.idle":"2024-12-21T03:13:16.284254Z","shell.execute_reply.started":"2024-12-21T03:13:16.277973Z","shell.execute_reply":"2024-12-21T03:13:16.283337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handling noise\ndef plotFeaturesKDE(df):\n    fig, axes = plt.subplots(20, 4, figsize=(25,50))\n    axes = axes.flatten()\n\n    for i, col in enumerate(df.columns):\n        sns.histplot(df[col], label=col, ax=axes[i])\n        \n    for j in range(len(df.columns), len(axes)):\n        fig.delaxes(axes[j])\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:18.192359Z","iopub.execute_input":"2024-12-21T03:13:18.192720Z","iopub.status.idle":"2024-12-21T03:13:18.198231Z","shell.execute_reply.started":"2024-12-21T03:13:18.192673Z","shell.execute_reply":"2024-12-21T03:13:18.197170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filtering out obvious false data like weight = 0\n# Physical weight\ntrain_data.loc[train_data['Physical-Weight']==0] # Drop\ntrain_data.drop(index=train_data.loc[train_data['Physical-Weight']==0].index, inplace=True)\n\ntrain_data.drop(index=train_data.loc[train_data['Physical-BMI']==0].index, inplace=True)\n\ntrain_data.drop(index=train_data.loc[train_data['BIA-BIA_BMI']==0].index, inplace=True)\n# train_data.loc[train_data['FGC-FGC_SRR']==0] # Keep because this indicates unhealthy participants","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T03:13:19.739291Z","iopub.execute_input":"2024-12-21T03:13:19.739640Z","iopub.status.idle":"2024-12-21T03:13:19.753075Z","shell.execute_reply.started":"2024-12-21T03:13:19.739612Z","shell.execute_reply":"2024-12-21T03:13:19.751777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handling skewed data\nskewed_data = ['Basic_Demos-Age',\n 'Basic_Demos-Sex',\n 'CGAS-CGAS_Score',\n 'Physical-BMI',\n 'Physical-Height',\n 'Physical-Weight',\n 'Physical-Diastolic_BP',\n 'Physical-HeartRate',\n 'Physical-Systolic_BP',\n 'FGC-FGC_CU',\n 'FGC-FGC_CU_Zone',\n 'FGC-FGC_PU',\n 'FGC-FGC_PU_Zone',\n 'FGC-FGC_SRL',\n 'FGC-FGC_SRL_Zone',\n 'FGC-FGC_SRR',\n 'FGC-FGC_SRR_Zone',\n 'FGC-FGC_TL',\n 'FGC-FGC_TL_Zone',\n 'BIA-BIA_Activity_Level_num',\n 'BIA-BIA_BMC',\n 'BIA-BIA_BMI',\n 'BIA-BIA_BMR',\n 'BIA-BIA_DEE',\n 'BIA-BIA_ECW',\n 'BIA-BIA_FFM',\n 'BIA-BIA_FFMI',\n 'BIA-BIA_FMI',\n 'BIA-BIA_Fat',\n 'BIA-BIA_Frame_num',\n 'BIA-BIA_ICW',\n 'BIA-BIA_LDM',\n 'BIA-BIA_LST',\n 'BIA-BIA_SMM',\n 'BIA-BIA_TBW',\n 'PAQ_A-PAQ_A_Total',\n 'PAQ_C-PAQ_C_Total',\n 'SDS-SDS_Total_Raw',\n 'SDS-SDS_Total_T',\n 'PreInt_EduHx-computerinternet_hoursday']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:23:56.167267Z","iopub.execute_input":"2024-12-19T16:23:56.167624Z","iopub.status.idle":"2024-12-19T16:23:56.174186Z","shell.execute_reply.started":"2024-12-19T16:23:56.167591Z","shell.execute_reply":"2024-12-19T16:23:56.172563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_cap = {}\n\nfor col in skewed_data:\n    feature_cap[col] = {\n        'lower': train_data[col].quantile(0.005),\n        'upper': train_data[col].quantile(0.995)\n    }\n\n    train_data[col] = train_data[col].clip(feature_cap[col]['lower'], feature_cap[col]['upper'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:23:56.175467Z","iopub.execute_input":"2024-12-19T16:23:56.175828Z","iopub.status.idle":"2024-12-19T16:23:56.231339Z","shell.execute_reply.started":"2024-12-19T16:23:56.175779Z","shell.execute_reply":"2024-12-19T16:23:56.230178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1>Feature engineering</h1>","metadata":{}},{"cell_type":"code","source":"bia_columns = [col for col in train_data.columns if col.split('-')[0]=='BIA']\nphysical_columsn = [col for col in train_data.columns if col.split('-')[0]=='Physical']\nfgc_columns = [col for col in train_data.columns if col.split('-')[0]=='FGC']\n\n# # Drop data with high missing ratio\n# for col in train_data.columns:\n#     percent_missing = np.round(len(train_data) - train_data[col].isnull().sum())\n#     print(f\"{col.ljust(30)} {str(percent_missing).rjust(6)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:23:56.232171Z","iopub.execute_input":"2024-12-19T16:23:56.232481Z","iopub.status.idle":"2024-12-19T16:23:56.240936Z","shell.execute_reply.started":"2024-12-19T16:23:56.232438Z","shell.execute_reply":"2024-12-19T16:23:56.240011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # BIA\n    # df['ECW_TBW_Ratio'] = df['BIA-BIA_ECW'] / df['BIA-BIA_TBW']\n    # df['ICW_TBW_Ratio'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    # df['Fat_Muscle_Ratio'] = df['BIA-BIA_Fat'] / df['BIA-BIA_SMM']\n    # df['BMI_Activity'] = df['BIA-BIA_BMI'] * df['BIA-BIA_Activity_Level_num']\n    # df['BMR_DEE_Interaction'] = df['BIA-BIA_BMR'] * df['BIA-BIA_DEE']\n    # df['TBW_Per_FFM'] = df['BIA-BIA_TBW'] / df['BIA-BIA_FFM']\n    # df['SMM_Per_FFM'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FFM']\n    \n    # # FGC\n    # df['healthy_zone'] = np.sum(df[['FGC-FGC_CU_Zone', 'FGC-FGC_PU_Zone', \n    #                                 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone']], axis=1)\n\n    # # Internet hour\n\n    # # Interacting features\n    # # Conver pounds to Kg\n    # df['Physical-Weight'] = df['Physical-Weight'] * 0.453592\n    # df['skeletal_muscle_ratio'] = df['BIA-BIA_SMM'] / df['Physical-Weight']\n    # df['lean_dry_ratio']  = df['BIA-BIA_LDM'] / df['Physical-Weight']\n    # df['lean_soft_ratio'] = df['BIA-BIA_LST'] / df['Physical-Weight']\n    # df['fat_free_ratio'] = df['BIA-BIA_FFM'] / df['Physical-Weight']\n    # df['mineral_ratio'] = df['BIA-BIA_BMC'] / 1000 / df['Physical-Weight'] # Gram to KG\n    # df['fat_ratio'] = df['BIA-BIA_Fat'] / 100\n\n    # df['internet_physical_bmi'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Physical-BMI']\n    # df['intertet_healthy_zone'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['healthy_zone']\n    # df['internet_bia_bmi']  = df['PreInt_EduHx-computerinternet_hoursday'] * df['BIA-BIA_BMI']\n    # df['internet_activity'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['BIA-BIA_Activity_Level_num']\n    \n    # # Activity questionaire\n    # df['PAQ_Total'] = df[['PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total']].mean(axis=1, skipna=True)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:23:56.241854Z","iopub.execute_input":"2024-12-19T16:23:56.245166Z","iopub.status.idle":"2024-12-19T16:23:56.263500Z","shell.execute_reply.started":"2024-12-19T16:23:56.245131Z","shell.execute_reply":"2024-12-19T16:23:56.261999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data[['Physical-Height', 'Physical-Weight', 'Physical-BMI']].info()\ntrain_data.loc[train_data['Physical-BMI'].isnull()] # Too many NaNs in these columns, the SII of these values are aslo 0 and 1 not 3\n# Remove all columns with missing Physical-BMI\ntrain_data.drop(index=train_data.loc[train_data['Physical-BMI'].isnull()].index, inplace=True)\n\ntrain_data = feature_engineering(train_data)\ntrain_data = train_data.drop(columns=pciat_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:23:56.264851Z","iopub.execute_input":"2024-12-19T16:23:56.265298Z","iopub.status.idle":"2024-12-19T16:23:56.333302Z","shell.execute_reply.started":"2024-12-19T16:23:56.265262Z","shell.execute_reply":"2024-12-19T16:23:56.331824Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1>Training</h1>","metadata":{}},{"cell_type":"code","source":"def transform(df):\n    df = df.drop(columns='id')\n\n    le = LabelEncoder()\n    le.fit(['Spring', 'Summer', 'Fall', 'Winter', 'missing'])\n    \n    season_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n           'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season',\n           'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n    \n    for col in season_cols:\n        df[col] = df[col].fillna('missing')\n        df[col] = le.transform(df[col])\n\n    high_miss_ratio = ['Physical-Waist_Circumference', 'Fitness_Endurance-Season',\n                       'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n                       'Fitness_Endurance-Time_Sec', 'FGC-FGC_GSND',\n                       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone']\n    \n    df.drop(columns=high_miss_ratio, inplace=True)\n\n\n    skewed_data = ['BIA-BIA_BMC', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', \n                   'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW']\n    \n    for col in skewed_data:\n        feature_cap[col] = {\n            'lower': df[col].quantile(0.005),\n            'upper': df[col].quantile(0.995)\n        }\n    \n        df[col] = df[col].clip(feature_cap[col]['lower'], feature_cap[col]['upper'])\n\n    df = feature_engineering(df)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:23:56.334591Z","iopub.execute_input":"2024-12-19T16:23:56.334959Z","iopub.status.idle":"2024-12-19T16:23:56.346365Z","shell.execute_reply.started":"2024-12-19T16:23:56.334923Z","shell.execute_reply":"2024-12-19T16:23:56.344672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom tqdm import tqdm\nfrom scipy.optimize import minimize\nfrom sklearn.base import clone\nfrom IPython.display import clear_output\nfrom sklearn.metrics import cohen_kappa_score, confusion_matrix\n\ndef getResult(model_for_validation, X_train, Y_train, X_val, Y_val):\n    fig, ax = plt.subplots(1, 2, figsize=(12,4))\n\n    sns.heatmap(confusion_matrix(Y_train, model_for_validation.predict(X_train)), annot=True, cmap='viridis', ax=ax[0])\n    print(f\"Train: {cohen_kappa_score(Y_train, model_for_validation.predict(X_train), weights='quadratic')}\")\n    \n    Y_pred = model_for_validation.predict(X_val)\n    sns.heatmap(confusion_matrix(Y_val, Y_pred), annot=True, cmap='viridis', ax=ax[1])\n    print(f\"Test: {cohen_kappa_score(Y_val, Y_pred, weights='quadratic')}\")\n    \ndef get_qwk(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\ndef threshold_Rounder(oof_non_rounded, thresholds, thresh_weights=[1,1,0.8]):\n    return np.where(oof_non_rounded < thresholds[0] * thresh_weights[0], 0,\n                    np.where(oof_non_rounded < thresholds[1] * thresh_weights[1], 1,\n                    np.where(oof_non_rounded < thresholds[2] * thresh_weights[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -get_qwk(y_true, rounded_p)\n\n\ndef trainModel(model_class, train_data, test_data): \n    X = train_data.drop(['sii'], axis=1)\n    Y = train_data['sii']\n\n    print(Y.value_counts())\n\n    folds = 5\n    cross_val = StratifiedKFold(n_splits=folds, shuffle=True, random_state=42)\n\n    train_score = []\n    val_score = []\n\n    oof_raw = np.zeros(len(Y), dtype=float)\n    oof_rounded = np.zeros(len(Y), dtype=int)\n    test_preds = np.zeros((len(test_data), folds))\n\n    imputer = KNNImputer(n_neighbors=5)\n    \n    for fold, (train_idx, val_idx) in enumerate(tqdm(cross_val.split(X, Y), desc=\"Training Folds\", total=folds)):\n        x_train, x_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = Y.iloc[train_idx], Y.iloc[val_idx]\n\n        x_train = pd.DataFrame(imputer.fit_transform(x_train), columns=x_train.columns)\n        x_val = pd.DataFrame(imputer.transform(x_val), columns=x_val.columns)\n    \n        model = clone(model_class)\n        model.fit(x_train, y_train)\n\n        y_train_pred = model.predict(x_train)\n        y_val_pred = model.predict(x_val)\n\n        oof_raw[val_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[val_idx] = y_val_pred_rounded\n\n        train_kappa = get_qwk(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = get_qwk(y_val, y_val_pred_rounded)\n\n        train_score.append(train_kappa)\n        val_score.append(val_kappa)\n\n        test_data = pd.DataFrame(imputer.transform(test_data), columns=test_data.columns)\n        test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_score):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(val_score):.4f}\")\n\n    kappaOptimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(Y, oof_raw), \n                              method='Nelder-Mead')\n    \n    assert kappaOptimizer.success, \"Optimizer did not converge\"\n    thresholds = kappaOptimizer.x\n    \n    oof_tuned = threshold_Rounder(oof_raw, thresholds)\n    tKappa = get_qwk(Y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {tKappa:.3f}\")\n\n    tpm = test_preds.mean(axis=1)\n    \n    tp_rounded = threshold_Rounder(tpm, thresholds)\n\n    # val confusion matrix\n    plt.figure(figsize=(6,6))\n    sns.heatmap(confusion_matrix(Y, oof_tuned), annot=True, cmap='viridis')\n    return tp_rounded\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:23:56.350599Z","iopub.execute_input":"2024-12-19T16:23:56.351023Z","iopub.status.idle":"2024-12-19T16:23:56.488071Z","shell.execute_reply.started":"2024-12-19T16:23:56.350982Z","shell.execute_reply":"2024-12-19T16:23:56.487039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import VotingRegressor, RandomForestRegressor\nfrom xgboost import XGBRegressor\nimport lightgbm as lgb\nfrom sklearn.impute import KNNImputer\n\n# Define parameters for each model\nxgb_params = {\n    'learning_rate': 0.05,\n    'max_depth': 3,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1, \n    'reg_lambda': 5, \n    'random_state': 42\n}\n\nrf_params = {\n     'bootstrap': True,\n     'criterion': 'squared_error',\n     'max_depth': 8,\n     'max_features': None,\n     'min_samples_leaf': 5,\n     'min_samples_split': 5,\n     'n_estimators': 150,\n     'random_state': 42\n}\n\nlgb_params = {\n    'objective': 'regression',\n    'learning_rate': 0.05,\n    'n_estimators': 200,\n    'max_depth': 3,\n    'num_leaves': 31,\n    'random_state': 42\n}\n\n# Initialize the models\nrf_reg = RandomForestRegressor(**rf_params)\nxgb_reg = XGBRegressor(**xgb_params)\nlgb_reg = lgb.LGBMRegressor(**lgb_params)\n\n# Create a VotingRegressor with three models\nvoting_reg = VotingRegressor(estimators=[('rf', rf_reg), ('xgb', xgb_reg), ('lgb', lgb_reg)])\n\n# Load and prepare test data\ntest_data = pd.read_csv(working_dir + 'test.csv')\nid_col = test_data['id']\ntest_data = transform(test_data)\n\n# Train the VotingRegressor\nresult = trainModel(voting_reg, train_data, test_data)\n\n# Prepare and save the output\noutput = pd.DataFrame({\n    'id': id_col,\n    'sii': result\n})\n\noutput.to_csv(output_dir + 'submission.csv', sep=',', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T16:24:45.016778Z","iopub.execute_input":"2024-12-19T16:24:45.017148Z","iopub.status.idle":"2024-12-19T16:25:23.128022Z","shell.execute_reply.started":"2024-12-19T16:24:45.017122Z","shell.execute_reply":"2024-12-19T16:25:23.126763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}